diff --git a/.gitignore b/.gitignore index 0c5c69ed3c..f245622118 100644 --- a/.gitignore +++ b/.gitignore @@ -2,6 +2,7 @@ CLAUDE.md Makefile docs/.claude/ +docs/flagos_homepage/.claude/ docs/_build/ docs/dockerfile make.bat @@ -12,6 +13,10 @@ to-do/ .playwright-mcp/ dockerfile dockerfile.bak +docs/flagos-homepage/AGENTS.md +docs/flagos-homepage/TODO.md +__pycache__/ +*.py[cod] # Playwright related files docs/flagcicd_zh/.playwright-mcp/ @@ -19,3 +24,4 @@ docs/flagcicd_zh/static/node_modules/ docs/flagcicd_zh/static/package.json docs/flagcicd_zh/static/package-lock.json docs/flagcicd_zh/static/*.js + diff --git a/docs/_ext/flagos_page_tags.py b/docs/_ext/flagos_page_tags.py new file mode 100644 index 0000000000..f1b1e88a43 --- /dev/null +++ b/docs/_ext/flagos_page_tags.py @@ -0,0 +1,125 @@ +"""Project-specific Sphinx helpers for the FlagOS homepage.""" + +import json + +from sphinx.addnodes import only, toctree +from sphinx.transforms.post_transforms import SphinxPostTransform + + +def _frontmatter_tags(app, docname): + """Return Sphinx tags declared in MyST frontmatter for one document.""" + metadata = app.env.metadata.get(docname, {}) + raw_tags = metadata.get("tags", []) + + if isinstance(raw_tags, str): + try: + raw_tags = json.loads(raw_tags) + except json.JSONDecodeError: + raw_tags = [raw_tags] + + if isinstance(raw_tags, str): + raw_tags = [raw_tags] + + return [tag for tag in raw_tags if isinstance(tag, str)] + + +def _eval_condition_with_page_tags(app, condition, page_tags): + """Evaluate an ``only`` expression with temporary page-level tags.""" + added_tags = [] + for tag in page_tags: + if not app.tags.has(tag): + app.tags.add(tag) + added_tags.append(tag) + + app.tags._condition_cache.clear() + try: + return app.tags.eval_condition(condition) + finally: + for tag in added_tags: + app.tags.remove(tag) + app.tags._condition_cache.clear() + + +class PageFrontmatterOnlyTransform(SphinxPostTransform): + """Apply ``only`` directives using tags from the current MyST page.""" + + default_priority = 49 + + def run(self, **kwargs): + docname = self.env.docname + page_tags = _frontmatter_tags(self.app, docname) + if not page_tags: + return + + for node in list(self.document.findall(only)): + try: + keep = _eval_condition_with_page_tags(self.app, node.get("expr", ""), page_tags) + except Exception: + continue + + if keep: + node.replace_self(node.children) + else: + node.parent.remove(node) + + +def _prune_inactive_only_toctrees(app, doctree): + """Keep toctree relations in sync with active ``only`` branches. + + Sphinx records ``toctree`` relationships while parsing a document, before + the ``only`` transform removes inactive branches. Rebuild the document's + include list from toctree nodes whose enclosing ``only`` conditions are + active. Page-level MyST ``tags`` frontmatter is treated as active while + evaluating the current document. + """ + docname = app.env.temp_data.get("docname") + if not docname or docname not in app.env.toctree_includes: + return + + page_tags = _frontmatter_tags(app, docname) + active_includes = [] + found_conditional_toctree = False + + for toctree_node in doctree.findall(toctree): + current = toctree_node.parent + keep = True + is_conditional = False + + while current is not None: + if isinstance(current, only): + is_conditional = True + expr = current.get("expr", "") + try: + if not _eval_condition_with_page_tags(app, expr, page_tags): + keep = False + break + except Exception: + # Let Sphinx's own only transform report malformed expressions. + pass + current = current.parent + + if is_conditional: + found_conditional_toctree = True + + if keep: + active_includes.extend(toctree_node.get("includefiles", [])) + + if found_conditional_toctree: + app.env.toctree_includes[docname] = active_includes + + compact_toc = app.env.tocs.get(docname) + if compact_toc is not None: + for only_node in list(compact_toc.findall(only)): + expr = only_node.get("expr", "") + try: + keep = _eval_condition_with_page_tags(app, expr, page_tags) + except Exception: + continue + if not keep and only_node.parent is not None: + only_node.parent.remove(only_node) + + +def setup(app): + app.add_post_transform(PageFrontmatterOnlyTransform) + app.connect("doctree-read", _prune_inactive_only_toctrees) + return {"version": "0.1", "parallel_read_safe": True, "parallel_write_safe": True} diff --git a/docs/conf.py b/docs/conf.py index 47e73707b5..df4c66edcb 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -109,7 +109,8 @@ def get_project(projects): "sphinx_tippy", "sphinxcontrib.lightbox2", # click-to-enlarge / lightbox for images "sphinx_tippy", - "sphinx_togglebutton" + "sphinx_togglebutton", + "flagos_page_tags" ] # Check and add actually installed extensions @@ -205,6 +206,13 @@ def get_project(projects): "html_title": "FlagGems-sglang Documentation", }, }, + "flaggems_sglang_zh": { + "use_config_file": False, + "config": { + "project": "FlagGems-sglang 文档中心", + "html_title": "FlagGems-sglang 文档中心", + }, + }, "flagdnn_en": { "use_config_file": False, "config": { @@ -648,7 +656,14 @@ def get_project(projects): # release = version # Exclude patterns - exclude all other project directories -exclude_patterns = ["_build", "shared", "_includes"] +exclude_patterns = [ + "_build", + "shared", + "_includes", + "chip_adaptation_guide/_shared", + "chip_adaptation_guide/TODO.md", + "chip_adaptation_guide_toctree_backup", +] all_projects = list(multiproject_projects.keys()) for project in all_projects: if project != docset: @@ -675,6 +690,8 @@ def get_project(projects): intersphinx_disabled_reftypes = ["*"] +myst_frontmatter_process = "yaml" + myst_enable_extensions = [ "dollarmath", "amsmath", @@ -753,6 +770,8 @@ def get_project(projects): # Common static paths html_static_path = ["_static", f"{docset}/_static"] html_css_files = ["custom.css", "homepage.css"] +if docset == "flagos_homepage": + html_css_files.append("guide.css") html_js_files = [] # html_logo = "img/logo.png" @@ -794,15 +813,18 @@ def get_project(projects): "use_download_button": False, "repository_url": "https://github.com/flagos-ai/KernelGen", "use_repository_button": True, - "secondary_sidebar_items": {}, + "secondary_sidebar_items": { + "**": ["page-toc"], + "flagos_homepage/index": [], + }, + "show_toc_level": 2, "footer_start": ["copyright"], "footer_end": [], "show_sphinx": False, "navbar_end": ["navbar-icon-links"] } - # Update secondary sidebar items for flagos_homepage - html_theme_options["secondary_sidebar_items"]["flagos_homepage/index"] = [] + # Keep the FlagOS homepage clean while enabling page-local TOC elsewhere. # html_sidebars is only for PyData Sphinx Theme html_sidebars = {} @@ -904,4 +926,3 @@ def get_project(projects): } suppress_warnings = ["epub.unknown_project_files"] - diff --git a/docs/flagcx_en/CHANGELOG.md b/docs/flagcx_en/CHANGELOG.md index 6e72ea6675..89e5b4f1ad 100644 --- a/docs/flagcx_en/CHANGELOG.md +++ b/docs/flagcx_en/CHANGELOG.md @@ -1,6 +1,16 @@ # Release Notes -- **[2026/06]** Released [v0.13](https://github.com/flagos-ai/FlagCX/releases/tag/v0.13.0-rc2.post1): +- **[2026/09]** Released [v.0.14.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.14.0): + + - Adds SHMEM Device API support with `USE_SHMEM` and `SHMEM_HOME` build controls. + - Adds Scalar IR and Unified IR Device API test coverage, including split intra-node and inter-node test targets. + - Adds ACCL/Barex network adaptor support and PPU backend integration controls. + - Extends PyTorch plugin build detection for PPU and the canonical `USE_ILUVATAR` backend flag. + - Updates Iluvatar build flag documentation: use `USE_ILUVATAR`; `USE_ILUVATAR_COREX` remains only as a deprecated compatibility alias. + - Widens the lane-mask ABI to 64 bits for Device IR paths. + - Updates packaging and dependency handling, including the NCCL 2.27 documentation floor for NVIDIA/NCCL wrapper usage. + +- **[2026/06]** Released [v.0.13.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.13.0): - Introduces FlagCX P2P Engine for one-sided RDMA operations, designed for integration with transfer frameworks like NIXL. - Adds IBRC P2P adaptor for InfiniBand-based P2P communication. @@ -11,7 +21,7 @@ - Adds new one-sided RDMA operations: `flagcxPut`, `flagcxBatchPut`, `flagcxReadCounter`, `flagcxWaitCounter`. - Optimizes RMA proxy with batched one-sided PUT operations for improved RDMA throughput. -- **[2026/05]** Released [v0.12](https://github.com/flagos-ai/FlagCX/releases/tag/v0.12.0): +- **[2026/05]** Released [v.0.12.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.12.0): - Adds support for Sunrise AI accelerators, including device adaptor `ptpuAdaptor` and CCL adaptor `pcclAdaptor`. - Extends PyTorch plugin support to PCCL backend. @@ -23,32 +33,32 @@ - Adds bootstrap extension for enhanced rendezvous capabilities. - Replaces C++17 features with C++11 equivalents for improved compatibility. -- **[2026/03]** Released [v0.11](https://github.com/flagos-ai/FlagCX/releases/tag/v0.11.0): +- **[2026/03]** Released [v.0.11.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.11.0): - Enables kernel-based communication on heterogeneous platforms, including NVIDIA and Hygon. - Adds support for both host-side and device-side one-sided communication semantics. - Introduces adaptor plugin support, enabling dynamic loading of user-defined Device, CCL, and Net adaptor implementations. -- **[2026/02]** Released [v0.10](https://github.com/flagos-ai/FlagCX/releases/tag/v0.10.0): +- **[2026/02]** Released [v.0.10.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.10.0): - Implements 11 chip-decoupled collective communication algorithms in uniRunner mode. - Refactors Device Intra-/Inter-node API and integrates NCCL Device API support on NVIDIA platforms. - Enhances usability with pip install support for FlagCX and an NCCL wrapper plugin for seamless adoption on NVIDIA platforms. -- **[2026/01]** Released [v0.9](https://github.com/flagos-ai/FlagCX/releases/tag/v0.9.0): +- **[2026/01]** Released [v.0.9.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.9.0): - Adds support for Enflame, including `topsAdaptor` and `ecclAdaptor`. - Extends flagcxCCLAdaptor to support symmetric operations. - Introduces the NCCL Device API in ncclAdaptor to enable customized AllReduce operations. - Refactors `glooAdaptor` to support both TCP and IB transports, with automatic NIC detection. -- **[2025/12]** Released [v0.8](https://github.com/flagos-ai/FlagCX/releases/tag/v0.8.0): +- **[2025/12]** Released [v.0.8.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.8.0): - Enables intra-node zero-copy to improve data transfer efficiency for small messages. - Supports a naive AllReduce implementation in uniRunner mode using a CPU-centric, device-assisted algorithm. - Adds one-sided communication primitives via the new APIs flagcxHeteroPut and flagcxHeteroPutSignal. -- **[2025/11]** Released [v0.7](https://github.com/flagos-ai/FlagCX/releases/tag/v0.7.0): +- **[2025/11]** Released [v.0.7.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.7.0): - Added support to TsingMicro, including device adaptor `tsmicroAdaptor` and CCL adaptor `tcclAdaptor`. - Implemented an experimental kernel-free non-reduce collective communication @@ -58,13 +68,13 @@ *AllReduce*, *AllGather*, *ReduceScatter*, and *AlltoAll*. - Enhanced `flagcxNetAdaptor` with one-sided primitives (`put`, `putSignal`, `waitValue`) and added retransmission support for reliability improvement. -- **[2025/10]** Released [v0.6](https://github.com/flagos-ai/FlagCX/releases/tag/v0.6.0): +- **[2025/10]** Released [v.0.6.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.6.0): - Implemented device-buffer IPC communication to support intra-node *SendRecv* operations. - Introduced _device-initiated, host-launched device-side primitives_, enabling kernel-based communication directly from devices. - Enhanced auto-tuning with 50% performance improvement on MetaX platforms for the *AllReduce* operations. -- **[2025/09]** Released [v0.5](https://github.com/flagos-ai/FlagCX/releases/tag/v0.5.0): +- **[2025/09]** Released [v.0.5.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.5.0): - Added support for AMD GPUs, including a device adaptor `hipAdaptor` and a CCL adaptor `rcclAdaptor`. - Introduced `flagcxNetAdaptor` to unify network backends, currently supporting socket, IBRC, UCX and IBUC (experimental). @@ -72,7 +82,7 @@ - Supported auto-tuning in homogeneous scenarios via `flagcxTuner`. - Added test automation in CI/CD for PyTorch APIs. -- **[2025/08]** Released [v0.4](https://github.com/flagos-ai/FlagCX/releases/tag/v0.4.0): +- **[2025/08]** Released [v.0.4.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.4.0): - Supported heterogeneous training of ERNIE4.5 (Baidu) on NVIDIA and Iluvatar GPUs with Paddle + FlagCX. - Improved heterogeneous communication across arbitrary NIC configurations, @@ -82,20 +92,20 @@ - Added an InterOp-level DSL to enable customized C2C algorithm design. - Provided user documentation under `docs/`. -- **[2025/07]** Released [v0.3](https://github.com/flagos-ai/FlagCX/releases/tag/v0.3.0): +- **[2025/07]** Released [v.0.3.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.3.0): - Integrated three additional native communication libraries: HCCL (Huawei), MUSACCL (Moore Threads) and MPI. - Enhanced heterogeneous collective communication operations with pipeline optimizations. - Introduced _device-side_ functions to enable device-buffer RDMA, complementing the existing _host-side_ functions. - Delivered a full-stack open-source solution, FlagScale + FlagCX, for efficient heterogeneous prefilling-decoding disaggregation. -- **[2025/05]** Released [v0.2](https://github.com/flagos-ai/FlagCX/releases/tag/v0.2.0): +- **[2025/05]** Released [v.0.2.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.2.0): - Integrated 3 additional native communications libraries, including MCCL (Moore Threads), XCCL (Mellanox) and DUCCL (BAAI). - Improved 11 heterogeneous collective communication operations with automatic topology detection and full support to single-NIC and multi-NIC environments. -- **[2025/04]** Released [v0.1](https://github.com/flagos-ai/FlagCX/releases/tag/v0.1.0): +- **[2025/04]** Released [v.0.1.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.1.0): - Added 5 native communications libraries including CCL adaptors for NCCL (NVIDIA), IXCCL (Iluvatar), and CNCL (Cambricon), @@ -104,5 +114,3 @@ - Provided a full-stack open-source solution, FlagScale + FlagCX, for efficient heterogeneous training. - Natively integrated into PaddlePaddle [v3.0.0](https://github.com/PaddlePaddle/Paddle/tree/v3.0.0), with support for both dynamic and static graphs. - - diff --git a/docs/flagcx_en/build.md b/docs/flagcx_en/build.md index f1d69f2191..46e00591f3 100644 --- a/docs/flagcx_en/build.md +++ b/docs/flagcx_en/build.md @@ -21,9 +21,11 @@ pip install . -v --no-build-isolation ```shell make =1 -j$(nproc) ``` + where `` is one of: - `USE_NVIDIA`: NVIDIA GPU support -- `USE_ILUVATAR_COREX`: Iluvatar Corex support +- `USE_ILUVATAR`: Iluvatar GPU support +- `USE_ILUVATAR_COREX`: deprecated compatibility alias for `USE_ILUVATAR` - `USE_CAMBRICON`: Cambricon support - `USE_METAX`: MetaX support - `USE_MUSA`: Moore Threads support @@ -34,11 +36,23 @@ where `` is one of: - `USE_TSM`: TsingMicro support - `USE_ENFLAME`: Enflame support - `USE_SUNRISE`: Sunrise AI support +- `USE_PPU`: PPU backend support - `USE_GLOO`: GLOO support - `USE_MPI`: MPI support -Note that Option A also supports `=1`, allowing users to explicitly specify the backend. Otherwise, it will be selected automatically. +Device API and integration build controls include: + +- `USE_SHMEM=1`: enable the SHMEM Device API adaptor. +- `SHMEM_HOME`: SHMEM installation path; defaults to `/usr/local/nvshmem`. +- `USE_ACCL_BAREX=1`: enable the ACCL/Barex network adaptor. +- `COMPILE_KERNEL=1`: compile kernel-enabled Device API components and tests. +- `HOST_CXX_STANDARD`: override the host C++ language standard when required by the build environment. +- `HOST_CXXFLAGS`: add host compiler flags. +- `JSON_INCLUDE_DIR`: override the JSON header include directory. + +The PPU integration commonly uses `USE_PPU=1 USE_ACCL_BAREX=1`; vendor runtime and transport settings are integration-specific and are not universal defaults. + +Option A also supports `=1`, allowing users to explicitly specify the backend. Otherwise, it will be selected automatically. -The default installation path is set to `build/`, you can manually set `BUILDDIR` environment variable to customize the build path. -You may also specify `DEVICE_HOME` and/or `CCL_HOME` to indicate the installation paths of the device runtime and installation path -of the communication libraries respectively. +The default installation path is set to `build/`; you can manually set the `BUILDDIR` environment variable to customize the build path. +You may also specify `DEVICE_HOME` and/or `CCL_HOME` to indicate the installation paths of the device runtime and communication libraries, respectively. diff --git a/docs/flagcx_en/environment-variables.md b/docs/flagcx_en/environment-variables.md index 7d6c4bc9cf..1f0d69366c 100644 --- a/docs/flagcx_en/environment-variables.md +++ b/docs/flagcx_en/environment-variables.md @@ -20,6 +20,7 @@ This document provides a comprehensive reference for all environment variables u - [Socket Network](#socket-network) - [UCX Network](#ucx-network) - [Gloo Network](#gloo-network) + - [ACCL/Barex and PPU Integration](#acclbarex-and-ppu-integration) - [Plugin Configuration](#plugin-configuration) - [Miscellaneous](#miscellaneous) - [Notes](#notes) @@ -105,6 +106,18 @@ This document provides a comprehensive reference for all environment variables u **Note**: These variables configure the FlagCX P2P Engine for one-sided RDMA operations, primarily used when integrating with transfer frameworks like NIXL. +### ACCL/Barex transport + +Environment variables change runtime behavior without rebuilding the library. The following settings apply only to deployments that build and use the ACCL/Barex transport adaptor: + +| Variable | Description | +|----------|-------------| +| `FLAGCX_P2P_TRANSPORT` | Set to `accl` to select the ACCL/Barex path when the adaptor is available. | +| `FLAGCX_ACCL_MAX_MR_MB` | Limits ACCL/Barex registered-memory chunking. `0` disables this limit. | +| `FLAGCX_VMM_ENABLE` | The current ACCL/Barex integration path requires `0`. | + +Build the adaptor with `USE_ACCL_BAREX=1`. These are deployment-specific settings, not universal defaults. + --- ## Topology Configuration @@ -224,7 +237,11 @@ This document provides a comprehensive reference for all environment variables u | Variable | Default | Description | |----------|---------|-------------| -| `FLAGCX_GLOO_IB_DISABLE` | 0 | When set to 1, disables IB for Gloo transport | +| `FLAGCX_GLOO_IB_DISABLE` | 0 | When set to 1, disables IB support for Gloo transport | + +### ACCL/Barex and PPU Integration + +Build the ACCL/Barex network adaptor with `USE_ACCL_BAREX=1`. The PPU integration commonly uses `USE_PPU=1 USE_ACCL_BAREX=1`; integration environments may also select `FLAGCX_P2P_TRANSPORT=accl`, enable `FLAGCX_MEM_ENABLE=1`, and disable virtual memory with `FLAGCX_VMM_ENABLE=0`. These settings are integration prerequisites, not universal defaults. --- diff --git a/docs/flagcx_en/getting-started.md b/docs/flagcx_en/getting-started.md index bbb1cece89..d6a3fa05ec 100644 --- a/docs/flagcx_en/getting-started.md +++ b/docs/flagcx_en/getting-started.md @@ -73,7 +73,7 @@ sudo docker run -itd \ make USE_SUNRISE=1 -j$(nproc) # Sunrise AI Platform ``` - See [](build.md) for the full list of supported backend flags. + See [Build and Installation](build.md) for the full list of supported backend flags. 2. Successful Build Result @@ -134,7 +134,7 @@ sudo docker run -itd \ ### Device API test Device API tests verify intra-node and inter-node communication via the Device API. -See [](testing.md) for the full list of test binaries, build instructions, and run examples. +See [Tests](testing.md) for the full list of test binaries, build instructions, and run examples. ### Torch API test @@ -281,4 +281,4 @@ See [](testing.md) for the full list of test binaries, build instructions, and r - On each host, compile and install the FlagCX communication API separately. - - Refer to the section [](#homogeneous-testing-with-flagcx) for detailed steps. + - Refer to the section [Homogeneous testing with FlagCX](#homogeneous-testing-with-flagcx) for detailed steps. diff --git a/docs/flagcx_en/paddle/README.md b/docs/flagcx_en/paddle/README.md index f5d1fc2162..a729cbf3bb 100644 --- a/docs/flagcx_en/paddle/README.md +++ b/docs/flagcx_en/paddle/README.md @@ -10,9 +10,9 @@ Train on a single type of hardware platform: | Hardware | User Guide | |:---------------:|:----------| -| Nvidia GPU | [](nvidia.md) | -| KLX XPU | [](kunlun.md) | -| Iluvatar GPU | [](iluvatar.md) | +| Nvidia GPU | [NVIDIA](nvidia.md) | +| KLX XPU | [Kunlunxin](kunlun.md) | +| Iluvatar GPU | [Iluvatar](iluvatar.md) | ## Heterogeneous training @@ -20,7 +20,7 @@ Train across **different hardware platforms** simultaneously: | Hardware Combination | User Guide | |:----------------------------:|:----------| -| Nvidia GPU + Iluvatar GPU | [](nvidia-iluvatar-hetero-train.md) | +| Nvidia GPU + Iluvatar GPU | [NVIDIA + Iluvatar](nvidia-iluvatar-hetero-train.md) | ```{toctree} :maxdepth: 3 diff --git a/docs/flagcx_en/paddle/nvidia-iluvatar-hetero-train.md b/docs/flagcx_en/paddle/nvidia-iluvatar-hetero-train.md index b55ec9ae51..60cff33d9e 100644 --- a/docs/flagcx_en/paddle/nvidia-iluvatar-hetero-train.md +++ b/docs/flagcx_en/paddle/nvidia-iluvatar-hetero-train.md @@ -2,7 +2,7 @@ ## Environment setup -Please refer to [](nvidia.md) and [](iluvatar.md) for environment setup and compiling Paddle with FlagCX on Nvidia and Iluvatar machines. +Please refer to [NVIDIA](nvidia.md) and [Iluvatar](iluvatar.md) for environment setup and compiling Paddle with FlagCX on Nvidia and Iluvatar machines. ## Training on heterogeneous ai accelerators (nvidia GPU + iluvatar GPU) diff --git a/docs/flagcx_en/testing.md b/docs/flagcx_en/testing.md index 0240d9fd0c..9baee83275 100644 --- a/docs/flagcx_en/testing.md +++ b/docs/flagcx_en/testing.md @@ -11,7 +11,7 @@ Performance tests are maintained in `test/perf/`, organized by API level: ```shell cd test/perf/host_api -make [USE_NVIDIA | USE_ILUVATAR_COREX | USE_CAMBRICON | USE_METAX | USE_MUSA | USE_KUNLUNXIN | USE_DU | USE_ASCEND | USE_AMD | USE_TSM | USE_ENFLAME | USE_SUNRISE]=1 +make [USE_NVIDIA | USE_ILUVATAR | USE_CAMBRICON | USE_METAX | USE_MUSA | USE_KUNLUNXIN | USE_DU | USE_ASCEND | USE_AMD | USE_TSM | USE_ENFLAME | USE_SUNRISE]=1 mpirun --allow-run-as-root -np 8 ./test_allreduce -b 128K -e 4G -f 2 ``` @@ -60,8 +60,13 @@ Device API tests are organized in two directories: | Binary | What it tests | |---|---| -| `test_device_api` | Correctness suite for 10 one-sided Device API kernels | -| `test_device_ir` | IR wrapper layer correctness | +| `test_device_api` | Correctness suite for one-sided Device API kernels | +| `test_device_api_intra` | Intra-node Device API correctness | +| `test_device_api_inter` | Inter-node Device API correctness | +| `test_device_ir_intra` | Intra-node Scalar IR correctness | +| `test_device_ir_inter` | Inter-node Scalar IR correctness | +| `test_device_ir_unified_intra` | Intra-node Unified IR correctness | +| `test_device_ir_unified_inter` | Inter-node Unified IR correctness | Build: diff --git a/docs/flagcx_en/user-guide.md b/docs/flagcx_en/user-guide.md index 6ff5f98b25..4699d40fe3 100644 --- a/docs/flagcx_en/user-guide.md +++ b/docs/flagcx_en/user-guide.md @@ -2,11 +2,11 @@ ## Environment configuration -Refer to the environment setup section in the [](getting-started.md) page. +Refer to the environment setup section in the [Getting Started](getting-started.md) page. ## Installation and compilation -Refer to [](getting-started.md) for FlagCX compilation and installation. +Refer to [Getting Started](getting-started.md) for FlagCX compilation and installation. ## API Reference @@ -71,7 +71,7 @@ The following API has been removed from the public interface: 1. Build and Installation - Refer to the Communication API test build and installation section in [](getting-started.md). + Refer to the Communication API test build and installation section in [Getting Started](getting-started.md). 2. Communication API Test @@ -142,7 +142,7 @@ The following API has been removed from the public interface: 1. Build and installation - Refer to [](getting-started.md) for instructions on building and installing the Torch API test. + Refer to [Getting Started](getting-started.md) for instructions on building and installing the Torch API test. 2. Torch API test execution @@ -198,7 +198,7 @@ The following API has been removed from the public interface: - `master_port`: Port used by the master node to establish the process group. All nodes must use the same port, and the port has to be available on all nodes. - `example.py`: Torch API test script. - - Refer to [](environment-variables.md) for the usage of the various `FLAGCX_XXX` environment variables. + - Refer to [Environment Variables](environment-variables.md) for the usage of the various `FLAGCX_XXX` environment variables. 3. Sample screenshot from a correct performance test @@ -210,7 +210,7 @@ The following steps shows an example in which we run the LLaMA3-8B model on Nvid 1. Build and installation - Refer to the Environment Setup and Build & Installation section in the [](getting-started.md) page. + Refer to the Environment Setup and Build & Installation section in the [Getting Started](getting-started.md) page. 2. Data preparation @@ -383,7 +383,7 @@ For kernel-based communication with Device API (available on NVIDIA and Hygon), export FLAGCX_MEM_ENABLE=1 ``` -Refer to [](environment-variables.md) for the full list of UniRunner-specific configuration variables (prefixed with `FLAGCX_UNIRUNNER_*`). +Refer to [Environment Variables](environment-variables.md) for the full list of UniRunner-specific configuration variables (prefixed with `FLAGCX_UNIRUNNER_*`). ### One-sided RDMA operations @@ -540,13 +540,13 @@ LD_PRELOAD=./build/lib/libnccl.so python your_training_script.py The wrapper intercepts NCCL API calls and routes them through FlagCX. A thread-local recursive guard prevents infinite recursion when FlagCX's internal NCCL adaptor calls back into NCCL. -Prerequisites: FlagCX built and installed, CUDA toolkit, real NCCL >= 2.21.0 (versions 2.21 through 2.27 supported). See `plugin/nccl/README.md` for full details. +Prerequisites: FlagCX built and installed, CUDA toolkit, and real NCCL >= 2.27. See `plugin/nccl/README.md` for full details. ### Communication API test 1. Build and Installation - Refer to the [](getting-started.md) documentation for instructions on + Refer to the [Getting Started](getting-started.md) documentation for instructions on environment setup, creating symbolic links, and how to build and install the software. 2. Verify MPICH Installation @@ -607,7 +607,7 @@ Prerequisites: FlagCX built and installed, CUDA toolkit, real NCCL >= 2.21.0 (ve /root/FlagCX/test/perf/test_allreduce -b 128K -e 4G -f 2 -w 5 -n 100 -p 1` ``` - - Refer to [](environment-variables.md) for the meaning and usage of `FLAGCX_XXX` environment variables. + - Refer to [Environment Variables](environment-variables.md) for the meaning and usage of `FLAGCX_XXX` environment variables. - **Note:** When using two GPUs per node in the heterogeneous Communication API test, some warnings may indicate that each node only has 1 GPU active. In this case, FlagCX will skip GPU-to-GPU AllReduce and fall back to host-based communication. diff --git a/docs/flagcx_zh/CHANGELOG.md b/docs/flagcx_zh/CHANGELOG.md index 1488b457f0..effbfa154b 100644 --- a/docs/flagcx_zh/CHANGELOG.md +++ b/docs/flagcx_zh/CHANGELOG.md @@ -1,6 +1,16 @@ # 发布说明 -- **[2026/06]** 发布 [v0.13](https://github.com/flagos-ai/FlagCX/releases/tag/v0.13.0-rc2.post1): +- **[2026/09]** 发布 [v.0.14.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.14.0): + + - 添加 SHMEM Device API 支持,并提供 `USE_SHMEM` 和 `SHMEM_HOME` 构建控制项。 + - 添加 Scalar IR 和 Unified IR 的 Device API 测试覆盖,包括拆分后的节点内和节点间测试目标。 + - 添加 ACCL/Barex 网络适配器支持和 PPU 后端集成控制项。 + - 扩展 PyTorch 插件构建检测,支持 PPU 和规范的 `USE_ILUVATAR` 后端标志。 + - 更新 Iluvatar 构建标志文档:新构建应使用 `USE_ILUVATAR`;`USE_ILUVATAR_COREX` 仅作为弃用的兼容别名保留。 + - 将 Device IR 路径中的 lane-mask ABI 扩展为 64 位。 + - 更新打包和依赖处理,包括 NVIDIA/NCCL 包装器使用场景下的 NCCL 2.27 文档版本下限。 + +- **[2026/06]** 发布 [v.0.13.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.13.0): - 引入 FlagCX P2P 引擎,用于单边 RDMA 操作,专为与 NIXL 等传输框架集成而设计。 - 添加 IBRC P2P 适配器,支持基于 InfiniBand 的 P2P 通信。 @@ -11,7 +21,7 @@ - 添加新的单边 RDMA 操作:`flagcxPut`、`flagcxBatchPut`、`flagcxReadCounter`、`flagcxWaitCounter`。 - 优化 RMA 代理,通过批量单边 PUT 操作提高 RDMA 吞吐量。 -- **[2026/05]** 发布 [v0.12](https://github.com/flagos-ai/FlagCX/releases/tag/v0.12.0): +- **[2026/05]** 发布 [v.0.12.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.12.0): - 添加 Sunrise AI 加速器支持,包括设备适配器 `ptpuAdaptor` 和 CCL 适配器 `pcclAdaptor`。 - 扩展 PyTorch 插件对 PCCL 后端的支持。 @@ -23,45 +33,45 @@ - 添加 bootstrap 扩展,增强汇合能力。 - 将 C++17 特性替换为 C++11 等效实现,提高兼容性。 -- **[2026/03]** 发布 [v0.11](https://github.com/flagos-ai/FlagCX/releases/tag/v0.11.0): +- **[2026/03]** 发布 [v.0.11.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.11.0): - 在异构平台上启用基于内核的通信,包括 NVIDIA 和 Hygon。 - 添加主机端和设备端单边通信语义支持。 - 引入适配器插件支持,支持动态加载用户定义的 Device、CCL 和 Net 适配器实现。 -- **[2026/02]** 发布 [v0.10](https://github.com/flagos-ai/FlagCX/releases/tag/v0.10.0): +- **[2026/02]** 发布 [v.0.10.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.10.0): - 在 uniRunner 模式下实现 11 个芯片解耦的集合通信算法。 - 重构设备节点内/节点间 API,并在 NVIDIA 平台上集成 NCCL Device API 支持。 - 增强易用性,支持 pip 安装 FlagCX,并提供 NCCL 包装插件以便在 NVIDIA 平台上无缝采用。 -- **[2026/01]** 发布 [v0.9](https://github.com/flagos-ai/FlagCX/releases/tag/v0.9.0): +- **[2026/01]** 发布 [v.0.9.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.9.0): - 添加 Enflame 支持,包括 `topsAdaptor` 和 `ecclAdaptor`。 - 扩展 flagcxCCLAdaptor 以支持对称操作。 - 在 ncclAdaptor 中引入 NCCL Device API,支持自定义 AllReduce 操作。 - 重构 `glooAdaptor`,支持 TCP 和 IB 传输,并自动检测网卡。 -- **[2025/12]** 发布 [v0.8](https://github.com/flagos-ai/FlagCX/releases/tag/v0.8.0): +- **[2025/12]** 发布 [v.0.8.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.8.0): - 启用节点内零拷贝,提高小消息数据传输效率。 - 在 uniRunner 模式下支持朴素的 AllReduce 实现,采用以 CPU 为中心、设备辅助的算法。 - 通过新 API flagcxHeteroPut 和 flagcxHeteroPutSignal 添加单边通信原语。 -- **[2025/11]** 发布 [v0.7](https://github.com/flagos-ai/FlagCX/releases/tag/v0.7.0): +- **[2025/11]** 发布 [v.0.7.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.7.0): - 添加 TsingMicro 支持,包括设备适配器 `tsmicroAdaptor` 和 CCL 适配器 `tcclAdaptor`。 - 实现实验性的无内核非归约集合通信(*SendRecv*、*AlltoAll*、*AlltoAllv*、*Broadcast*、*Gather*、*Scatter*、*AllGather*),使用设备缓冲区 IPC/RDMA。 - 在 NVIDIA、MetaX 和 Hygon 平台上启用自动调优,*AllReduce*、*AllGather*、*ReduceScatter* 和 *AlltoAll* 性能提升 1.02×–1.26×。 - 增强 `flagcxNetAdaptor`,添加单边原语(`put`、`putSignal`、`waitValue`)和重传支持,提高可靠性。 -- **[2025/10]** 发布 [v0.6](https://github.com/flagos-ai/FlagCX/releases/tag/v0.6.0): +- **[2025/10]** 发布 [v.0.6.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.6.0): - 实现设备缓冲区 IPC 通信,支持节点内 *SendRecv* 操作。 - 引入设备发起、主机启动的设备端原语,支持直接从设备进行基于内核的通信。 - 增强自动调优,MetaX 平台上 *AllReduce* 操作性能提升 50%。 -- **[2025/09]** 发布 [v0.5](https://github.com/flagos-ai/FlagCX/releases/tag/v0.5.0): +- **[2025/09]** 发布 [v.0.5.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.5.0): - 添加 AMD GPU 支持,包括设备适配器 `hipAdaptor` 和 CCL 适配器 `rcclAdaptor`。 - 引入 `flagcxNetAdaptor` 统一网络后端,目前支持 socket、IBRC、UCX 和 IBUC(实验性)。 @@ -69,7 +79,7 @@ - 通过 `flagcxTuner` 支持同构场景下的自动调优。 - 在 CI/CD 中添加 PyTorch API 测试自动化。 -- **[2025/08]** 发布 [v0.4](https://github.com/flagos-ai/FlagCX/releases/tag/v0.4.0): +- **[2025/08]** 发布 [v.0.4.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.4.0): - 支持 ERNIE4.5(百度)在 NVIDIA 和 Iluvatar GPU 上使用 Paddle + FlagCX 进行异构训练。 - 改进任意网卡配置下的异构通信,部署更加稳健灵活。 @@ -77,19 +87,19 @@ - 添加 InterOp 级 DSL,支持自定义 C2C 算法设计。 - 在 `docs/` 下提供用户文档。 -- **[2025/07]** 发布 [v0.3](https://github.com/flagos-ai/FlagCX/releases/tag/v0.3.0): +- **[2025/07]** 发布 [v.0.3.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.3.0): - 集成三个额外的原生通信库:HCCL(华为)、MUSACCL(摩尔线程)和 MPI。 - 通过流水线优化增强异构集合通信操作。 - 引入设备端函数支持设备缓冲区 RDMA,补充现有的主机端函数。 - 提供全栈开源解决方案 FlagScale + FlagCX,实现高效的异构预填充-解码分离。 -- **[2025/05]** 发布 [v0.2](https://github.com/flagos-ai/FlagCX/releases/tag/v0.2.0): +- **[2025/05]** 发布 [v.0.2.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.2.0): - 集成 3 个额外的原生通信库,包括 MCCL(摩尔线程)、XCCL(Mellanox)和 DUCCL(BAAI)。 - 改进 11 个异构集合通信操作,支持自动拓扑检测和单网卡/多网卡环境。 -- **[2025/04]** 发布 [v0.1](https://github.com/flagos-ai/FlagCX/releases/tag/v0.1.0): +- **[2025/04]** 发布 [v.0.1.0](https://github.com/flagos-ai/FlagCX/releases/tag/v0.1.0): - 添加 5 个原生通信库,包括 NCCL(NVIDIA)、IXCCL(Iluvatar)和 CNCL(寒武纪)的 CCL 适配器,以及主机 CCL 适配器 GLOO 和 Bootstrap。 - 使用 C2C(集群到集群)算法支持 11 个异构集合通信操作。 diff --git a/docs/flagcx_zh/build.md b/docs/flagcx_zh/build.md index a28a43c718..dc27027f45 100644 --- a/docs/flagcx_zh/build.md +++ b/docs/flagcx_zh/build.md @@ -21,23 +21,38 @@ pip install . -v --no-build-isolation ```shell make =1 -j$(nproc) ``` + 其中 `` 为以下选项之一: -- `USE_NVIDIA`: NVIDIA GPU 支持 -- `USE_ILUVATAR_COREX`: Iluvatar Corex 支持 -- `USE_CAMBRICON`: 寒武纪支持 -- `USE_METAX`: MetaX 支持 -- `USE_MUSA`: 摩尔线程支持 -- `USE_KUNLUNXIN`: 昆仑芯支持 -- `USE_DU`: 海光支持 -- `USE_ASCEND`: 华为昇腾支持 -- `USE_AMD`: AMD 支持 -- `USE_TSM`: 清微智能支持 -- `USE_ENFLAME`: 燧原支持 -- `USE_SUNRISE`: Sunrise AI 支持 -- `USE_GLOO`: GLOO 支持 -- `USE_MPI`: MPI 支持 - -注意,方式 A 也支持 `=1`,允许用户显式指定后端。否则将自动选择。 - -默认安装路径设置为 `build/`,您可以手动设置 `BUILDDIR` 环境变量来自定义构建路径。 -您也可以指定 `DEVICE_HOME` 和/或 `CCL_HOME` 来指示设备运行时的安装路径和通信库的安装路径。 \ No newline at end of file +- `USE_NVIDIA`:NVIDIA GPU 支持 +- `USE_ILUVATAR`:Iluvatar GPU 支持 +- `USE_ILUVATAR_COREX`:`USE_ILUVATAR` 的弃用兼容别名 +- `USE_CAMBRICON`:寒武纪支持 +- `USE_METAX`:MetaX 支持 +- `USE_MUSA`:摩尔线程支持 +- `USE_KUNLUNXIN`:昆仑芯支持 +- `USE_DU`:海光支持 +- `USE_ASCEND`:华为昇腾支持 +- `USE_AMD`:AMD 支持 +- `USE_TSM`:清微智能支持 +- `USE_ENFLAME`:燧原支持 +- `USE_SUNRISE`:Sunrise AI 支持 +- `USE_PPU`:PPU 后端支持 +- `USE_GLOO`:GLOO 支持 +- `USE_MPI`:MPI 支持 + +Device API 和集成构建控制项包括: + +- `USE_SHMEM=1`:启用 SHMEM Device API 适配器。 +- `SHMEM_HOME`:SHMEM 安装路径,默认为 `/usr/local/nvshmem`。 +- `USE_ACCL_BAREX=1`:启用 ACCL/Barex 网络适配器。 +- `COMPILE_KERNEL=1`:编译启用内核的 Device API 组件和测试。 +- `HOST_CXX_STANDARD`:在构建环境需要时覆盖主机端 C++ 语言标准。 +- `HOST_CXXFLAGS`:添加主机端编译器选项。 +- `JSON_INCLUDE_DIR`:覆盖 JSON 头文件包含目录。 + +PPU 集成通常使用 `USE_PPU=1 USE_ACCL_BAREX=1`;厂商运行时和传输设置取决于具体集成环境,并非通用默认值。 + +方式 A 同样支持 `=1`,允许用户显式指定后端;否则系统会自动选择后端。 + +默认安装路径为 `build/`,您可以手动设置 `BUILDDIR` 环境变量来自定义构建路径。 +您也可以指定 `DEVICE_HOME` 和/或 `CCL_HOME`,分别指示设备运行时和通信库的安装路径。 diff --git a/docs/flagcx_zh/environment-variables.md b/docs/flagcx_zh/environment-variables.md index 3ed524f78e..396fab5e78 100644 --- a/docs/flagcx_zh/environment-variables.md +++ b/docs/flagcx_zh/environment-variables.md @@ -21,6 +21,7 @@ - [Socket 网络](#socket-网络) - [UCX 网络](#ucx-网络) - [Gloo 网络](#gloo-网络) + - [ACCL/Barex 和 PPU 集成](#acclbarex-和-ppu-集成) - [插件配置](#插件配置) - [其他](#其他) - [注意事项](#注意事项) @@ -227,6 +228,10 @@ |----------|---------|-------------| | `FLAGCX_GLOO_IB_DISABLE` | 0 | 设置为 1 时,为 Gloo 传输禁用 IB | +### ACCL/Barex 和 PPU 集成 + +使用 `USE_ACCL_BAREX=1` 构建 ACCL/Barex 网络适配器。PPU 集成通常使用 `USE_PPU=1 USE_ACCL_BAREX=1`;具体集成环境还可能选择 `FLAGCX_P2P_TRANSPORT=accl`、启用 `FLAGCX_MEM_ENABLE=1`,并通过 `FLAGCX_VMM_ENABLE=0` 禁用虚拟内存。这些设置是特定集成环境的前置条件,并非通用默认值。 + --- ## 插件配置 diff --git a/docs/flagcx_zh/getting-started.md b/docs/flagcx_zh/getting-started.md index c34ef915f4..8676626da9 100644 --- a/docs/flagcx_zh/getting-started.md +++ b/docs/flagcx_zh/getting-started.md @@ -73,7 +73,7 @@ sudo docker run -itd \ make USE_SUNRISE=1 -j$(nproc) # Sunrise AI 平台 ``` - 参见 [](build.md) 获取支持的后端标志完整列表。 + 参见[构建与安装](build.md)获取支持的后端标志完整列表。 2. 构建成功结果 @@ -134,7 +134,7 @@ sudo docker run -itd \ ### Device API 测试 Device API 测试验证通过 Device API 进行的节点内和节点间通信。 -参见 [](testing.md) 获取测试二进制文件的完整列表、构建说明和运行示例。 +参见[测试](testing.md)获取测试二进制文件的完整列表、构建说明和运行示例。 ### Torch API 测试 @@ -281,4 +281,4 @@ Device API 测试验证通过 Device API 进行的节点内和节点间通信。 - 在每台主机上分别编译安装 FlagCX 通信 API。 - - 参见 [](#使用-flagcx-进行同构测试) 部分的详细步骤。 + - 参见[使用 FlagCX 进行同构测试](getting-started.md)部分的详细步骤。 diff --git a/docs/flagcx_zh/paddle/README.md b/docs/flagcx_zh/paddle/README.md index 040867c5a9..2978ea8c89 100644 --- a/docs/flagcx_zh/paddle/README.md +++ b/docs/flagcx_zh/paddle/README.md @@ -10,9 +10,9 @@ FlagCX 现已作为**可选的高性能通信后端**完全集成到 Paddle 中 | 硬件 | 用户指南 | |:---------------:|:----------| -| NVIDIA GPU | [](nvidia.md) | -| 昆仑芯 XPU | [](kunlun.md) | -| Iluvatar GPU | [](iluvatar.md) | +| NVIDIA GPU | [NVIDIA](nvidia.md) | +| 昆仑芯 XPU | [昆仑芯](kunlun.md) | +| Iluvatar GPU | [Iluvatar](iluvatar.md) | ## 异构训练 @@ -20,7 +20,7 @@ FlagCX 现已作为**可选的高性能通信后端**完全集成到 Paddle 中 | 硬件组合 | 用户指南 | |:----------------------------:|:----------| -| NVIDIA GPU + Iluvatar GPU | [](nvidia-iluvatar-hetero-train.md) | +| NVIDIA GPU + Iluvatar GPU | [NVIDIA + Iluvatar](nvidia-iluvatar-hetero-train.md) | ```{toctree} :maxdepth: 3 diff --git a/docs/flagcx_zh/paddle/nvidia-iluvatar-hetero-train.md b/docs/flagcx_zh/paddle/nvidia-iluvatar-hetero-train.md index a710c04f94..0debefc645 100644 --- a/docs/flagcx_zh/paddle/nvidia-iluvatar-hetero-train.md +++ b/docs/flagcx_zh/paddle/nvidia-iluvatar-hetero-train.md @@ -2,7 +2,7 @@ ## 环境配置 -请参考 [](nvidia.md) 和 [](iluvatar.md) 了解在 NVIDIA 和天数智芯机器上的环境配置以及使用 FlagCX 编译 Paddle 的方法。 +请参考 [NVIDIA](nvidia.md) 和 [Iluvatar](iluvatar.md),了解在 NVIDIA 和天数智芯机器上的环境配置以及使用 FlagCX 编译 Paddle 的方法。 ## 在异构 AI 加速器上训练(NVIDIA GPU + 天数智芯 GPU) diff --git a/docs/flagcx_zh/testing.md b/docs/flagcx_zh/testing.md index 5445df3688..cd953b7737 100644 --- a/docs/flagcx_zh/testing.md +++ b/docs/flagcx_zh/testing.md @@ -11,7 +11,7 @@ ```shell cd test/perf/host_api -make [USE_NVIDIA | USE_ILUVATAR_COREX | USE_CAMBRICON | USE_METAX | USE_MUSA | USE_KUNLUNXIN | USE_DU | USE_ASCEND | USE_AMD | USE_TSM | USE_ENFLAME | USE_SUNRISE]=1 +make [USE_NVIDIA | USE_ILUVATAR | USE_CAMBRICON | USE_METAX | USE_MUSA | USE_KUNLUNXIN | USE_DU | USE_ASCEND | USE_AMD | USE_TSM | USE_ENFLAME | USE_SUNRISE]=1 mpirun --allow-run-as-root -np 8 ./test_allreduce -b 128K -e 4G -f 2 ``` @@ -60,8 +60,13 @@ Device API 测试组织在两个目录中: | 二进制文件 | 测试内容 | |---|---| -| `test_device_api` | 10 个单边 Device API 内核的正确性测试套件 | -| `test_device_ir` | IR 包装层正确性测试 | +| `test_device_api` | 单边 Device API 内核的正确性测试套件 | +| `test_device_api_intra` | 节点内 Device API 正确性测试 | +| `test_device_api_inter` | 节点间 Device API 正确性测试 | +| `test_device_ir_intra` | 节点内 Scalar IR 正确性测试 | +| `test_device_ir_inter` | 节点间 Scalar IR 正确性测试 | +| `test_device_ir_unified_intra` | 节点内 Unified IR 正确性测试 | +| `test_device_ir_unified_inter` | 节点间 Unified IR 正确性测试 | 构建: diff --git a/docs/flagcx_zh/user-guide.md b/docs/flagcx_zh/user-guide.md index 7f0421ae8a..7157422671 100644 --- a/docs/flagcx_zh/user-guide.md +++ b/docs/flagcx_zh/user-guide.md @@ -2,13 +2,21 @@ ## 环境配置 -参见 [](getting-started.md) 页面中的环境配置部分。 +参见[快速开始](getting-started.md)页面中的环境配置部分。 ## 安装与编译 -参见 [](getting-started.md) 了解 FlagCX 编译和安装。 +参见[快速开始](getting-started.md)了解 FlagCX 编译和安装。 -## API 参考 +## 面向初学者的关键概念 + +- **Device API(设备端 API)**:用于从设备端内核发起或配合通信的较低层接口。与主要由 CPU 侧程序调用的 Host API 相比,Device API 更接近设备内存、设备内核和底层通信路径。相关测试通常需要使用 `COMPILE_KERNEL=1` 构建 FlagCX。 +- **内存注册**:在通信前向 FlagCX 声明一个缓冲区,使通信运行时能够检查并使用该缓冲区。注册完成后,应按照对应 API 的生命周期调用注销或释放操作。 +- **窗口注册**:把一段已分配的内存注册为可重复使用的通信窗口,常用于单边通信和 Device API 路径。使用窗口时,应配对调用 `flagcxCommWindowRegister` 与 `flagcxCommWindowDeregister`,并通过匹配的内存分配/释放 API 管理缓冲区。 +- **P2P(点对点通信)引擎**:负责建立对等连接、注册内存并执行 RDMA 读写;它可以向传输框架提供单边数据传输能力,但不是集合通信 API。 +- **RMA / 单边通信**:允许一方读取或写入远端已注册内存,远端不必像传统双边通信那样同时调用对应的数据传输函数。FlagCX 的单边操作包括 `Get`、`Put` 和信号/计数器同步。 +- **PD(Prefill/Decode 分离)**:在大模型推理中,将处理输入提示词的 Prefill 阶段与生成输出 token 的 Decode 阶段拆分到不同服务或工作进程。FlagCX 可作为这类异构部署中的通信组件;PD 本身不是 FlagCX 的一个 API。 +- **PTD(Prefill-Transfer-Decode)**:用于观察或分析 Prefill、其中间数据传输以及 Decode 三个阶段的性能流程。PTD 是 profiling/可观测性工作流,不是 FlagCX 通信 API;具体指标和可视化由相应的部署工具链提供。 ### 设备句柄管理 @@ -71,7 +79,7 @@ flagcxResult_t flagcxGetUniqueId(flagcxUniqueId_t uniqueId); 1. 构建与安装 - 参见 [](getting-started.md) 中的通信 API 测试构建与安装部分。 + 参见[快速开始](getting-started.md)中的通信 API 测试构建与安装部分。 2. 通信 API 测试 @@ -138,7 +146,7 @@ flagcxResult_t flagcxGetUniqueId(flagcxUniqueId_t uniqueId); 1. 构建与安装 - 参见 [](getting-started.md) 了解 Torch API 测试的构建与安装说明。 + 参见[快速开始](getting-started.md)了解 Torch API 测试的构建与安装说明。 2. Torch API 测试执行 @@ -194,7 +202,7 @@ flagcxResult_t flagcxGetUniqueId(flagcxUniqueId_t uniqueId); - `master_port`:主节点用于建立进程组的端口。 所有节点必须使用相同的端口,且该端口在所有节点上必须可用。 - `example.py`:Torch API 测试脚本。 - - 参见 [](environment-variables.md) 了解各种 `FLAGCX_XXX` 环境变量的用法。 + - 参见[环境变量](environment-variables.md)了解各种 `FLAGCX_XXX` 环境变量的用法。 3. 正确性能测试的示例截图 @@ -206,7 +214,7 @@ flagcxResult_t flagcxGetUniqueId(flagcxUniqueId_t uniqueId); 1. 构建与安装 - 参见 [](getting-started.md) 页面中的环境配置和构建安装部分。 + 参见[快速开始](getting-started.md)页面中的环境配置和构建安装部分。 2. 数据准备 @@ -377,7 +385,7 @@ UniRunner 支持异构硬件上的所有标准集合操作(AllReduce、AllGath export FLAGCX_MEM_ENABLE=1 ``` -参见 [](environment-variables.md) 获取 UniRunner 特定配置变量的完整列表(前缀为 `FLAGCX_UNIRUNNER_*`)。 +参见[环境变量](environment-variables.md)获取 UniRunner 特定配置变量的完整列表(前缀为 `FLAGCX_UNIRUNNER_*`)。 ### 单边 RDMA 操作 @@ -534,13 +542,13 @@ LD_PRELOAD=./build/lib/libnccl.so python your_training_script.py 该包装器拦截 NCCL API 调用并通过 FlagCX 路由它们。线程本地递归保护可防止 FlagCX 内部 NCCL 适配器回调 NCCL 时出现无限递归。 -前置条件:FlagCX 已构建并安装、CUDA 工具包、真实 NCCL >= 2.21.0(支持 2.21 到 2.27 版本)。详见 `plugin/nccl/README.md`。 +前置条件:FlagCX 已构建并安装、CUDA 工具包以及真实 NCCL >= 2.27。详见 `plugin/nccl/README.md`。 ### 通信 API 测试 1. 构建与安装 - 参见 [](getting-started.md) 文档,了解环境配置、创建符号链接以及如何构建和安装软件的说明。 + 参见[快速开始](getting-started.md)文档,了解环境配置、创建符号链接以及如何构建和安装软件的说明。 2. 验证 MPICH 安装 @@ -598,7 +606,7 @@ LD_PRELOAD=./build/lib/libnccl.so python your_training_script.py /root/FlagCX/test/perf/test_allreduce -b 128K -e 4G -f 2 -w 5 -n 100 -p 1` ``` - - 参见 [](environment-variables.md) 了解 `FLAGCX_XXX` 环境变量的含义和用法。 + - 参见[环境变量](environment-variables.md)了解 `FLAGCX_XXX` 环境变量的含义和用法。 - **注意:** 在异构通信 API 测试中,当每个节点使用 2 个 GPU 时,某些警告可能指示每个节点只有 1 个 GPU 处于活动状态。在这种情况下,FlagCX 将跳过 GPU 到 GPU 的 AllReduce,回退到基于主机的通信。 diff --git a/docs/flaggems_sglang_en/getting_started/install.md b/docs/flaggems_sglang_en/getting_started/install.md index fecff7cd13..58d4eefe74 100644 --- a/docs/flaggems_sglang_en/getting_started/install.md +++ b/docs/flaggems_sglang_en/getting_started/install.md @@ -5,7 +5,7 @@ For a fresh installation of FlagGems-sglang, follow the steps below. 1. Install build dependencies. ```{code-block} bash - pip install -U scikit-build-core>=0.11 pybind11 ninja cmake + pip install -U 'scikit-build-core>=0.11' pybind11 ninja cmake ``` 2. Clone and install FlagGems-sglang. @@ -15,3 +15,27 @@ For a fresh installation of FlagGems-sglang, follow the steps below. cd FlagGems-sglang pip install . ``` + + For development, use editable installation: + + ```{code-block} bash + pip install --no-build-isolation -e . + ``` + + To run the test suites, install the test dependencies as well: + + ```{code-block} bash + pip install -e '.[test]' + ``` + +3. Optionally select the backend explicitly. The runtime detects the device on import; set `DNN_VENDOR` to force a specific backend instead: + + ```{code-block} bash + export DNN_VENDOR=nvidia + ``` + +## Verify the installation + +```{code-block} bash +python -c "import flaggems_sglang; print(flaggems_sglang.device, flaggems_sglang.vendor_name); print(len(flaggems_sglang.all_registered_ops()), 'operators')" +``` diff --git a/docs/flaggems_sglang_en/getting_started/requirements.md b/docs/flaggems_sglang_en/getting_started/requirements.md index 1534f0d142..6a343e2e5a 100644 --- a/docs/flaggems_sglang_en/getting_started/requirements.md +++ b/docs/flaggems_sglang_en/getting_started/requirements.md @@ -5,19 +5,32 @@ Before installing FlagGems-sglang, ensure your environment meets the following r ## Software requirements - **Python** 3.8 or higher -- **PyTorch** with CUDA support +- **PyTorch** 2.6.0 or higher, with support for your accelerator - **Triton** compatible with your GPU architecture +- **PyYAML** and **SQLAlchemy**, required by the operator registry at runtime - **pip** for package installation ## Build dependencies The following packages are required to build FlagGems-sglang from source: +- `setuptools` >= 64.0 - `scikit-build-core` >= 0.11 - `pybind11` -- `ninja` -- `cmake` + +`ninja` and `cmake` are only needed when a vendor extension is compiled, which is why the installation guide installs them up front. + +## Test dependencies + +The `test` extra installs what the test and benchmark suites need: + +- `pytest` >= 7.1.0 +- `numpy` >= 1.26 +- `scipy` >= 1.14 +- `cupy-cuda12x` + +`cupy-cuda12x` targets CUDA devices. On other accelerators, install the CuPy wheel matching that platform. ## Hardware requirements -- A CUDA-capable GPU is required for running most operators. +A supported accelerator is required to run most operators. The test suite can compare against a CPU reference with `--ref cpu` for the operators that support it. diff --git a/docs/flaggems_sglang_en/index.md b/docs/flaggems_sglang_en/index.md index a80401fb1a..5e413dee81 100644 --- a/docs/flaggems_sglang_en/index.md +++ b/docs/flaggems_sglang_en/index.md @@ -51,6 +51,16 @@ Run tests and benchmarks to validate and measure operator performance. [Learn more »](user_guide/run-tests-and-benchmark.md) ::: +:::{grid-item-card} {octicon}`list-unordered;1.5em;sd-mr-1` Operator List +:link: reference/operator_list +:link-type: doc + +The 40 operators exported by FlagGems-sglang, grouped by operator family. + ++++ +[Learn more »](reference/operator_list.md) +::: + :::: --- diff --git a/docs/flaggems_sglang_en/overview/features.md b/docs/flaggems_sglang_en/overview/features.md index cda06fbfb7..929c8c5e89 100644 --- a/docs/flaggems_sglang_en/overview/features.md +++ b/docs/flaggems_sglang_en/overview/features.md @@ -4,5 +4,34 @@ FlagGems-sglang provides the following key features: - **Operators have undergone deep performance tuning** — Each operator is carefully optimized for throughput and latency across multiple hardware backends. - **Triton kernel call optimization** — Kernel launch overhead is minimized through specialized Triton kernel patterns and autotuning. -- **Flexible multi-backend support mechanism** — The library supports a variety of GPU hardware platforms, allowing operators to run efficiently regardless of the underlying device. -- **Support for common SGLang operators** — Includes optimized implementations of operators frequently used in SGLang inference, such as `relu`, `outer`, and flashinfer-related operators. +- **Flexible multi-backend support mechanism** — Operator implementations are resolved at import time by a three-level registrar (generic, vendor, architecture), so the same call site selects the best available kernel for the device in use. +- **Support for common SGLang operators** — Includes optimized implementations of operators frequently used in SGLang inference, such as `silu_and_mul`, `apply_token_bitmask`, `fused_rmsnorm`, and the Mamba and Mixture-of-Experts families. + +## Supported backends + +Vendors ship their specializations under `src/flaggems_sglang/runtime/backend/_/`. Each vendor folder declares the device it serves: + +| Vendor folder | Device | Specialized operators | +|---------------|--------|-----------------------| +| `_kunlunxin` | `cuda` | 37 | +| `_ascend` | `npu` | 33 | +| `_enflame` | `gcu` | 32 | +| `_iluvatar` | `cuda` | 29 | +| `_metax` | `cuda` | 27 | +| `_hygon` | `cuda` | 22 | +| `_mthreads` | `musa` | 2 | +| `_nvidia` | `cuda` | 1, plus `hopper/` and `ampere/` architecture folders | +| `_thead` | `cuda` | descriptor only | +| `_amd` | `cuda` | descriptor only | + +A vendor that does not specialize an operator still runs it: routing falls back to the generic Triton implementation in `flaggems_sglang.ops`. + +The device is detected at import time from a per-vendor query command: `nvidia-smi`, `npu-smi info`, `mx-smi`, `ixsmi`, `hy-smi`, `efsmi -L`, `xpu-smi`, `rocm-smi`, `mthreads-gmi`, `ppu-smi`. Setting the `DNN_VENDOR` environment variable overrides detection and forces a specific backend, which is how the test suites are pointed at a particular vendor. + +## Relationship with FlagGems and sglang-plugin-FL + +- **FlagGems**: the general-purpose operator library. It replaces ATen operators globally in PyTorch through `flag_gems.enable()`. +- **FlagGems-sglang**: this repository. It provides SGLang-oriented operator implementations together with the tests and benchmarks that validate them, exposed through the `flaggems_sglang` Python package. +- **sglang-plugin-FL**: the SGLang plugin layer. It hooks SGLang's operator dispatch and routes framework calls to the selected backend implementation, using FlagGems for general ATen coverage and FlagGems-sglang operators for the SGLang-specific fused kernels. + +Vendors bringing up a new backend should start from the bring-up guide kept in the upstream repository at `src/flaggems_sglang/runtime/backend/README.md`. diff --git a/docs/flaggems_sglang_en/overview/overview.md b/docs/flaggems_sglang_en/overview/overview.md index d525b0a072..c88067c2f8 100644 --- a/docs/flaggems_sglang_en/overview/overview.md +++ b/docs/flaggems_sglang_en/overview/overview.md @@ -1,13 +1,38 @@ # FlagGems-sglang Overview -FlagGems-sglang is part of [FlagOS](https://flagos.io/). FlagGems-sglang is a high-performance operator library designed for multiple hardware backends. It provides optimized implementations of common SGLang operators and supports high-performance inference and deployment for a variety of widely used models. +FlagGems-sglang is part of [FlagOS](https://flagos.io/). It is a high-performance operator library designed for multiple hardware backends. It provides optimized implementations of common SGLang operators and supports high-performance inference and deployment for a variety of widely used models. FlagGems-sglang is a high-performance deep learning operator library implemented using the [Triton programming language](https://github.com/openai/triton) launched by OpenAI. By integrating with SGLang, FlagGems-sglang accelerates inference workloads through optimized Triton kernels that replace default operator implementations, delivering significant performance gains across diverse hardware platforms. +The library ships 40 generic operators. They cover activation and gating, attention, Mamba and SSM chunks and scans, chunked cumulative sums, Mixture-of-Experts routing and reduction, LoRA projections, normalization, INT8 quantization, rotary positional embedding, sampling and speculative decoding. See [Operator List](../reference/operator_list.md) for the full set. + +## Multi-level operator routing + +Every operator is resolved for the current device by a three-level registrar; later levels override earlier ones on a name collision: + +| Priority | Source | Purpose | +|----------|--------|---------| +| 0 | `flaggems_sglang.ops` | Generic Triton implementation | +| 1 | `flaggems_sglang.runtime.backend._/ops` | Per-vendor specialization | +| 2 | `flaggems_sglang.runtime.backend._//ops` | Per-architecture specialization, for example `_nvidia/hopper/ops` | + +Routing is driven by function name: a vendor override defines `def my_op(...)` in `_/ops/my_op.py` and lists it in that module's `__all__`. Operators a vendor does not specialize automatically fall back to the generic implementation, so each vendor ships only the kernels it actually adapts. + +The resolved implementation is attached to the package namespace, so callers use one entry point regardless of the underlying hardware: + +```python +import flaggems_sglang + +flaggems_sglang.device # device name, for example cuda +flaggems_sglang.vendor_name # detected vendor, for example nvidia +flaggems_sglang.all_registered_ops() # operator names resolved for this device +flaggems_sglang.get_op("silu_and_mul") # resolved callable +flaggems_sglang.silu_and_mul.__module__ # module the resolved implementation came from +``` + ```{toctree} -:maxdepth: 2 features.md diff --git a/docs/flaggems_sglang_en/reference/operator_list.md b/docs/flaggems_sglang_en/reference/operator_list.md new file mode 100644 index 0000000000..14a0b59c04 --- /dev/null +++ b/docs/flaggems_sglang_en/reference/operator_list.md @@ -0,0 +1,104 @@ +# Operator List + +This page lists the operators exported by FlagGems-sglang, sourced from `conf/operators.yaml` and the `__all__` of `src/flaggems_sglang/ops/*.py`. + +The generic operator set contains 40 operators, and the two sources agree name for name. Each operator is implemented in Triton and resolved for the current device by the three-level registrar described in [Overview](../overview/overview.md). + +## Activation and Gating + +| Operator | Description | +|----------|-------------| +| `gelu_and_mul` | Fused GELU gate and multiply operation. Computes gelu(x[..., :d]) * x[..., d:] in a single pass. Stride-aware so non-contiguous gate/up splits are handled without a copy. | +| `silu_and_mul` | Fused SiLU gate and multiply operation. Computes silu(x[..., :d]) * x[..., d:] in a single pass. Optimized for both decode and prefill phases with adaptive tiling strategies. | +| `silu_and_mul_masked` | Masked SiLU gate and multiply for grouped Mixture-of-Experts activations. Computes only the valid token rows indicated by the per-group masked length. | + +## Attention + +| Operator | Description | +|----------|-------------| +| `context_attention` | Prefill/context-stage scaled dot-product attention on packed variable-length sequences. Supports causal masking over sequences delimited by per-batch start offsets and lengths. | +| `decode_attention` | Decode-stage attention against a paged KV cache. Gathers keys and values through an indirection table for single-token queries. | +| `decode_grouped_attention` | Grouped-query variant of decode-stage paged attention. Shares each KV head across a group of query heads to cut cache bandwidth. | +| `merge_state` | Merges attention states from prefix and suffix sequences using log-sum-exp scaling. Used for continuous batching and chunked attention computation. | + +## Mamba and SSM + +| Operator | Description | +|----------|-------------| +| `bmm_chunk` | Chunked batched matrix multiplication for the Mamba2 SSD scan. Computes per-chunk C @ B^T products with optional causal masking. | +| `causal_conv1d_fn` | Generic depthwise causal 1-D convolution over continuous-batched sequences. Computes depthwise causal convolution with optional bias and SiLU activation. | +| `chunk_state` | Mamba2 per-chunk SSM hidden-state accumulation. Accumulates x against the decay- and dt-scaled state-projection matrix per chunk. | +| `chunk_state_varlen` | Variable-length variant of the Mamba2 per-chunk state accumulation. Handles packed sequences delimited by cumulative sequence lengths. | +| `selective_state_update` | Single-step Mamba selective state-space update. Advances the recurrent state in float32 with optional softplus dt, D skip and gating. | +| `state_passing` | Mamba2 SSD inter-chunk state-passing scan. Sequentially propagates each chunk's final state forward from the initial state. | +| `mamba_layernorm_gated` | Gated layer/RMS normalization for Mamba blocks. Applies grouped normalization with a SiLU gate before or after the norm. | + +## Chunked Cumulative Sum + +| Operator | Description | +|----------|-------------| +| `chunk_cumsum` | Chunk-wise cumulative sum of the Mamba timestep and state-decay terms. Computes dt with optional softplus and bias, plus the cumulative dA per chunk. | +| `chunk_local_cumsum_scalar` | Computes local cumulative sum over chunked sequences with optional scaling and reverse direction. Supports both direct and transposed cumsum computation. | +| `chunk_local_cumsum_vector` | Within-chunk cumulative sum of a per-token, per-head vector gate. Supports optional scaling and reverse-direction accumulation. | + +## Mixture of Experts (MoE) + +| Operator | Description | +|----------|-------------| +| `fused_moe_gemm` | Fused Mixture-of-Experts GEMM operation that combines token routing and expert computation. Performs weighted expert-specific matrix multiplication for MoE models. | +| `fused_moe_router_cudacore` | Fused Mixture-of-Experts router evaluated on CUDA cores. Computes router logits with optional softcapping and correction bias, then selects top-k experts. | +| `fused_moe_router_tensorcore` | Tensor Core variant of the fused Mixture-of-Experts router. Mathematically identical to the CUDA-core router, tuned for Tensor Core throughput. | +| `moe_fused_gate` | Fused Mixture-of-Experts gating with expert-group selection. Scores experts, applies bias and top-k selection, then optionally renormalizes and rescales. | +| `moe_fused_mul_sum` | Fused weighted reduction of Mixture-of-Experts outputs. Scales each top-k expert output by its routing weight and sums over top-k, masking slots dropped by expert-parallel routing. | +| `moe_sum_reduce` | Sums the per-expert outputs of a Mixture-of-Experts layer. Reduces over the top-k dimension and applies the routed scaling factor. | +| `per_group_transpose` | Per-expert group transpose operation for MoE models. Transposes tensor blocks grouped by expert assignment for efficient computation. | +| `sigmoid_gate_topk_renorm` | Sigmoid expert gating with top-k selection and weight renormalization. Applies optional bias, shared-expert handling, and route/global scaling. | + +## LoRA + +| Operator | Description | +|----------|-------------| +| `embedding_lora_a` | LoRA A-projection over an embedding lookup for segmented multi-adapter batches. Selects the per-segment adapter weight and projects the gathered rows. | +| `gate_up_lora_b` | LoRA B-projection for fused gate/up projections in segmented multi-adapter batches. Expands the low-rank activations and accumulates onto the base output. | +| `qkv_lora_b` | LoRA B-projection for fused QKV projections in segmented multi-adapter batches. Expands the low-rank activations into per-slice Q, K and V offsets of the base output. | +| `sgemm_lora_a` | LoRA A-projection GEMM for segmented multi-adapter batches. Shrinks activations to the adapter rank, optionally over a stack of projections. | +| `sgemm_lora_b` | LoRA B-projection GEMM for segmented multi-adapter batches. Expands the low-rank activations and accumulates onto the base output. | + +## Normalization + +| Operator | Description | +|----------|-------------| +| `fused_rmsnorm` | Fused root-mean-square layer normalization with a learned weight. Computes the reciprocal RMS and applies the scale in a single pass. | + +## Quantization + +| Operator | Description | +|----------|-------------| +| `per_token_group_quant_int8` | Per-token, per-group symmetric INT8 quantization. Emits quantized values alongside one scale per token group. | +| `per_token_quant_int8` | Per-token symmetric INT8 quantization. Emits quantized values alongside one scale per token row. | + +## Rotary Positional Embedding + +| Operator | Description | +|----------|-------------| +| `interleaved_rope` | Interleaved-layout Rotary Positional Embedding. Rotates adjacent element pairs using multi-section temporal, height, and width splits. | +| `mrope_fused` | Multi-dimensional Rotary Positional Embedding (mRoPE) for query and key tensors. Applies position-dependent rotation across temporal, height, and width dimensions. | +| `rotary_embedding` | Rotary Positional Embedding with precomputed cosine and sine tables. Supports both the interleaved and half-split rotation layouts. | + +## Sampling + +| Operator | Description | +|----------|-------------| +| `apply_token_bitmask` | Applies a packed bitmask to logits for grammar-constrained decoding. Masks out disallowed vocabulary tokens by setting them to -inf. | +| `softcap_inplace_logits` | In-place logit softcapping. Computes cap * tanh(x / cap) over the full logit tensor. Bounds logit magnitude before sampling without allocating an output buffer. | +| `softcap_out` | Out-of-place logit softcapping. Computes cap * tanh(x / cap) and returns an FP32 tensor. Uses a deterministic shape heuristic to pick a tiling on every launch. | + +## Speculative Decoding + +| Operator | Description | +|----------|-------------| +| `draft_topk1` | Top-1 draft token selection for speculative decoding. Picks the argmax draft token per position and writes it to the draft token column. | + +## Multi-backend operator coverage + +Beyond the generic implementations, vendors ship specializations under `src/flaggems_sglang/runtime/backend/_/ops/`. Per-vendor coverage counts are in the backends table in [Features](../overview/features.md). Operators a vendor does not specialize fall back to the generic Triton implementation. diff --git a/docs/flaggems_sglang_en/release_notes/release-notes.md b/docs/flaggems_sglang_en/release_notes/release-notes.md index 6666886f5f..9fd131c468 100644 --- a/docs/flaggems_sglang_en/release_notes/release-notes.md +++ b/docs/flaggems_sglang_en/release_notes/release-notes.md @@ -2,10 +2,10 @@ This section includes the release information for FlagGems-sglang. -## Initial release +## v0.1.0 -- **Added features**: - - Initial release of FlagGems-sglang as part of FlagOS. - - Provides high-performance Triton-based operator implementations for SGLang inference. - - Includes flexible multi-backend support mechanism for diverse GPU hardware. - - Deep performance tuning and Triton kernel call optimization. +**Added features**: + +- 40 operators implemented in Triton for SGLang inference, covering activation and gating, attention, Mamba/SSM, chunked cumulative sums, Mixture-of-Experts, LoRA projections, normalization, INT8 quantization, rotary positional embedding, sampling and speculative decoding. +- Flexible multi-backend support: operators are resolved through a three-level registrar (generic / vendor / architecture), with vendor specializations for KunlunXin, Ascend, Enflame, Iluvatar, MetaX, Hygon, MThreads and NVIDIA, plus Hopper and Ampere architecture folders for NVIDIA. +- Deep performance tuning and Triton kernel call optimization, with per-operator test and benchmark suites and accuracy recording. diff --git a/docs/flaggems_sglang_en/user_guide/run-tests-and-benchmark.md b/docs/flaggems_sglang_en/user_guide/run-tests-and-benchmark.md index 574c7491d1..3d9750b170 100644 --- a/docs/flaggems_sglang_en/user_guide/run-tests-and-benchmark.md +++ b/docs/flaggems_sglang_en/user_guide/run-tests-and-benchmark.md @@ -7,20 +7,34 @@ The following commands are verified in the FlagGems-sglang repository and can be ## Run tests ```bash -cd /workspace/FlagGems-sglang +cd FlagGems-sglang pytest -q tests --collect-only -pytest -q tests/test_outer.py --quick +pytest -q tests/test_silu_and_mul.py --quick +``` + +To check an operator against the CPU reference instead of the device reference: + +```bash +pytest -q tests/test_silu_and_mul.py --ref cpu --quick ``` ## Run benchmark ```bash -cd /workspace/FlagGems-sglang +cd FlagGems-sglang pytest -q benchmark --collect-only -pytest -q benchmark/test_outer.py::test_outer --level core --iter 1 --warmup 1 +pytest -q benchmark/test_silu_and_mul.py --level core --iter 1 --warmup 1 +``` + +The benchmark suite records accuracy alongside performance. Passing `--record log` writes a per-run log, and `--record json` writes `accuracy_result.json` instead. + +```bash +pytest -q benchmark/test_silu_and_mul.py --level core --record log ``` ```{note} -- Most tests/benchmarks require a CUDA-capable GPU runtime. +- Most tests/benchmarks require an accelerator runtime; the reference implementation can run on CPU with `--ref cpu`. - `--collect-only` is recommended first to quickly check import and discovery. +- To point the run at a specific vendor backend, set `DNN_VENDOR` before invoking pytest. +- `--level core` runs the core shape set; the default level is more comprehensive and takes longer. ``` diff --git a/docs/flaggems_sglang_en/user_guide/usage.md b/docs/flaggems_sglang_en/user_guide/usage.md index e1988514ac..ae88cf13bd 100644 --- a/docs/flaggems_sglang_en/user_guide/usage.md +++ b/docs/flaggems_sglang_en/user_guide/usage.md @@ -1,18 +1,30 @@ # Use operators -After installing FlagGems-sglang, you can use its optimized operators directly in your Python code. +After installing FlagGems-sglang, import the package and call the operators directly. Every operator resolved for the current device is attached to the package namespace, so the call site stays the same regardless of the underlying hardware. -For example, import the library and call the operators on CUDA tensors: +The example below uses two exported operators, `silu_and_mul` and `fused_rmsnorm`: ```python import torch import flaggems_sglang -# Create a tensor -x = torch.randn(1024, device='cuda') +# Create a tensor; hidden_states has shape [..., 2d] +x = torch.randn(1024, 4096, device=flaggems_sglang.device) -# Apply ReLU activation -y = flaggems_sglang.ops.relu(x) +# Gated SiLU: out = silu(x[..., :d]) * x[..., d:] +y = flaggems_sglang.silu_and_mul(x) + +# Fused RMSNorm over the last dimension +weight = torch.ones(2048, device=flaggems_sglang.device) +z = flaggems_sglang.fused_rmsnorm(y, weight, eps=1e-5) +``` + +To check which implementation was resolved on the current machine: + +```python +print(flaggems_sglang.device, flaggems_sglang.vendor_name) +print(flaggems_sglang.silu_and_mul.__module__) +print(len(flaggems_sglang.all_registered_ops()), "operators registered") ``` For a full operator list, see [Operator List](../reference/operator_list.md). diff --git a/docs/flaggems_sglang_zh/getting_started/getting-started.md b/docs/flaggems_sglang_zh/getting_started/getting-started.md new file mode 100644 index 0000000000..53b690d159 --- /dev/null +++ b/docs/flaggems_sglang_zh/getting_started/getting-started.md @@ -0,0 +1,10 @@ +# 快速入门 + +本节说明安装和运行 FlagGems-sglang 的环境要求,并指导您完成安装并使用其优化的算子。 + +```{toctree} +:maxdepth: 2 + +requirements.md +install.md +``` diff --git a/docs/flaggems_sglang_zh/getting_started/install.md b/docs/flaggems_sglang_zh/getting_started/install.md new file mode 100644 index 0000000000..0ce3a570fb --- /dev/null +++ b/docs/flaggems_sglang_zh/getting_started/install.md @@ -0,0 +1,41 @@ +# 安装 FlagGems-sglang + +如需全新安装 FlagGems-sglang,请按照以下步骤操作。 + +1. 安装构建依赖项。 + + ```{code-block} bash + pip install -U 'scikit-build-core>=0.11' pybind11 ninja cmake + ``` + +2. 克隆并安装 FlagGems-sglang。 + + ```{code-block} bash + git clone https://github.com/flagos-ai/FlagGems-sglang.git + cd FlagGems-sglang + pip install . + ``` + + 开发场景请使用可编辑安装方式: + + ```{code-block} bash + pip install --no-build-isolation -e . + ``` + + 如需运行测试套件,请同时安装测试依赖: + + ```{code-block} bash + pip install -e '.[test]' + ``` + +3. 可选:显式指定后端。运行时会在导入时自动检测设备;如需强制指定后端,请设置 `DNN_VENDOR`: + + ```{code-block} bash + export DNN_VENDOR=nvidia + ``` + +## 验证安装 + +```{code-block} bash +python -c "import flaggems_sglang; print(flaggems_sglang.device, flaggems_sglang.vendor_name); print(len(flaggems_sglang.all_registered_ops()), 'operators')" +``` diff --git a/docs/flaggems_sglang_zh/getting_started/requirements.md b/docs/flaggems_sglang_zh/getting_started/requirements.md new file mode 100644 index 0000000000..1a4bf34abf --- /dev/null +++ b/docs/flaggems_sglang_zh/getting_started/requirements.md @@ -0,0 +1,36 @@ +# 环境要求 + +安装 FlagGems-sglang 之前,请确认环境满足以下要求。 + +## 软件要求 + +- **Python** 3.8 及以上 +- **PyTorch** 2.6.0 及以上,且支持所使用的加速卡 +- **Triton** 与所用 GPU 架构兼容 +- **PyYAML** 与 **SQLAlchemy**,算子注册表在运行时依赖这两个包 +- **pip** 用于安装软件包 + +## 构建依赖 + +从源码构建 FlagGems-sglang 需要以下软件包: + +- `setuptools` >= 64.0 +- `scikit-build-core` >= 0.11 +- `pybind11` + +`ninja` 与 `cmake` 仅在编译厂商扩展时需要,因此安装指南会一并预先装好。 + +## 测试依赖 + +`test` extra 会安装测试与基准套件所需的软件包: + +- `pytest` >= 7.1.0 +- `numpy` >= 1.26 +- `scipy` >= 1.14 +- `cupy-cuda12x` + +`cupy-cuda12x` 面向 CUDA 设备;其他加速卡请安装与该平台匹配的 CuPy 轮子。 + +## 硬件要求 + +运行大多数算子需要受支持的加速卡。测试套件可对支持的算子使用 `--ref cpu` 与 CPU 参考实现做对比。 diff --git a/docs/flaggems_sglang_zh/index.md b/docs/flaggems_sglang_zh/index.md new file mode 100644 index 0000000000..cb2ee25428 --- /dev/null +++ b/docs/flaggems_sglang_zh/index.md @@ -0,0 +1,93 @@ +# FlagGems-sglang 文档 + +```{button-ref} getting_started/getting-started +:ref-type: myst +:color: primary +:class: sd-btn-lg sd-px-4 sd-py-2 sd-fw-bold + +快速入门 +``` + +::::{grid} 1 2 2 3 +:gutter: 1 1 1 2 + +:::{grid-item-card} {octicon}`browser;1.5em;sd-mr-1` 概览 +:link: overview/overview +:link-type: doc + +快速了解 FlagGems-sglang 以及一些基本概念。 + ++++ +[了解更多 »](overview/overview.md) +::: + +:::{grid-item-card} {octicon}`book;1.5em;sd-mr-1` 快速入门 +:link: getting_started/getting-started +:link-type: doc + +概述 FlagGems-sglang 的安装要求,并提供从安装到使用的分步说明。 + ++++ +[了解更多 »](getting_started/getting-started.md) +::: + +:::{grid-item-card} {octicon}`broadcast;1.5em;sd-mr-1` 用户指南 +:link: user_guide/user-guide +:link-type: doc + +指导您如何在 SGLang 推理工作流中使用 FlagGems-sglang 算子。 + ++++ +[了解更多 »](user_guide/user-guide.md) +::: + +:::{grid-item-card} {octicon}`beaker;1.5em;sd-mr-1` 测试与基准 +:link: user_guide/run-tests-and-benchmark +:link-type: doc + +运行测试和基准来验证并衡量算子性能。 + ++++ +[了解更多 »](user_guide/run-tests-and-benchmark.md) +::: + +:::{grid-item-card} {octicon}`list-unordered;1.5em;sd-mr-1` 算子列表 +:link: reference/operator_list +:link-type: doc + +FlagGems-sglang 导出的 40 个算子,按算子家族分组。 + ++++ +[了解更多 »](reference/operator_list.md) +::: + +:::: + +--- + +```{toctree} +:caption: 📑 发布说明 +:maxdepth: 5 +:hidden: + +release_notes/release-notes.md +``` + +```{toctree} +:caption: 📚 指南 +:maxdepth: 5 +:hidden: + +overview/overview.md +getting_started/getting-started.md +user_guide/user-guide.md +user_guide/run-tests-and-benchmark.md +``` + +```{toctree} +:caption: 📖 参考 +:maxdepth: 5 +:hidden: + +reference/operator_list.md +``` diff --git a/docs/flaggems_sglang_zh/overview/features.md b/docs/flaggems_sglang_zh/overview/features.md new file mode 100644 index 0000000000..0de76f760d --- /dev/null +++ b/docs/flaggems_sglang_zh/overview/features.md @@ -0,0 +1,37 @@ +# 特性 + +FlagGems-sglang 提供以下关键特性: + +- **算子经过深度性能调优** — 每个算子都针对多种硬件后端的吞吐量与延迟做了精心优化。 +- **Triton 内核调用优化** — 通过专门的 Triton 内核模式与自动调优,最大限度降低内核启动开销。 +- **灵活的多后端支持机制** — 算子实现在导入时由三级注册器(通用、厂商、架构)解析,因此同一处调用会为当前设备选择最合适的可用内核。 +- **支持常用 SGLang 算子** — 包含 SGLang 推理中高频算子的优化实现,例如 `silu_and_mul`、`apply_token_bitmask`、`fused_rmsnorm`,以及 Mamba 与混合专家家族的算子。 + +## 支持的后端 + +各厂商的特化实现位于 `src/flaggems_sglang/runtime/backend/_/`。每个厂商目录都声明了自己服务的设备: + +| 厂商目录 | 设备 | 特化算子数 | +|---------------|--------|-----------------------| +| `_kunlunxin` | `cuda` | 37 | +| `_ascend` | `npu` | 33 | +| `_enflame` | `gcu` | 32 | +| `_iluvatar` | `cuda` | 29 | +| `_metax` | `cuda` | 27 | +| `_hygon` | `cuda` | 22 | +| `_mthreads` | `musa` | 2 | +| `_nvidia` | `cuda` | 1,另有 `hopper/` 与 `ampere/` 架构目录 | +| `_thead` | `cuda` | 仅描述信息 | +| `_amd` | `cuda` | 仅描述信息 | + +厂商未特化的算子仍可运行:路由会回退到 `flaggems_sglang.ops` 中的通用 Triton 实现。 + +设备在导入时通过各厂商的查询命令自动检测:`nvidia-smi`、`npu-smi info`、`mx-smi`、`ixsmi`、`hy-smi`、`efsmi -L`、`xpu-smi`、`rocm-smi`、`mthreads-gmi`、`ppu-smi`。设置 `DNN_VENDOR` 环境变量可覆盖自动检测,强制指定后端,测试套件正是通过这种方式指向特定厂商。 + +## 与 FlagGems、sglang-plugin-FL 的关系 + +- **FlagGems**:通用算子库。它通过 `flag_gems.enable()` 在 PyTorch 中全局替换 ATen 算子。 +- **FlagGems-sglang**:即本仓库。它提供面向 SGLang 的算子实现以及用于验证的测试与基准,通过 `flaggems_sglang` Python 包对外暴露。 +- **sglang-plugin-FL**:SGLang 插件层。它挂钩 SGLang 的算子调度,将框架调用路由到选定的后端实现:通用 ATen 覆盖使用 FlagGems,SGLang 专用融合内核使用 FlagGems-sglang 的算子。 + +厂商接入新后端时,可从上游仓库的 `src/flaggems_sglang/runtime/backend/README.md` 开始,其中说明了目录结构与 `VendorDescriptor` 字段。 diff --git a/docs/flaggems_sglang_zh/overview/overview.md b/docs/flaggems_sglang_zh/overview/overview.md new file mode 100644 index 0000000000..2c046e1c55 --- /dev/null +++ b/docs/flaggems_sglang_zh/overview/overview.md @@ -0,0 +1,39 @@ +# FlagGems-sglang 概览 + +FlagGems-sglang 是 [FlagOS](https://flagos.io/) 的一部分,是一个面向多种硬件后端的高性能算子库。它提供了常见 SGLang 算子的优化实现,并支持多种广泛使用的模型进行高性能推理与部署。 + +FlagGems-sglang 是一个使用 OpenAI 推出的 [Triton 编程语言](https://github.com/openai/triton) 实现的高性能深度学习算子库。 + +通过与 SGLang 集成,FlagGems-sglang 以优化的 Triton 内核替代默认算子实现来加速推理负载,在多种硬件平台上带来显著的性能提升。 + +本算子库提供 40 个通用算子,覆盖激活与门控、注意力、Mamba 与 SSM 的分块与扫描、分块累积求和、混合专家的路由与归约、LoRA 投影、归一化、INT8 量化、旋转位置编码、采样与投机解码。完整列表见[算子列表](../reference/operator_list.md)。 + +## 多级算子路由 + +每个算子都会针对当前设备经三级注册器解析,同名冲突时后一级覆盖前一级: + +| 优先级 | 来源 | 用途 | +|----------|--------|---------| +| 0 | `flaggems_sglang.ops` | 通用 Triton 实现 | +| 1 | `flaggems_sglang.runtime.backend._/ops` | 面向厂商的特化实现 | +| 2 | `flaggems_sglang.runtime.backend._//ops` | 面向具体架构的特化实现,例如 `_nvidia/hopper/ops` | + +路由以函数名为依据:厂商特化只需在 `_/ops/my_op.py` 中定义 `def my_op(...)`,并在该模块的 `__all__` 中列出即可。厂商未特化的算子会自动回退到通用实现,因此每个厂商只需提供自己真正适配的内核。 + +解析后的实现直接挂载在包命名空间上,因此无论底层硬件是什么,调用方使用的入口都一致: + +```python +import flaggems_sglang + +flaggems_sglang.device # 设备名,例如 cuda +flaggems_sglang.vendor_name # 检测到的厂商,例如 nvidia +flaggems_sglang.all_registered_ops() # 当前设备解析出的算子名 +flaggems_sglang.get_op("silu_and_mul") # 解析后的可调用对象 +flaggems_sglang.silu_and_mul.__module__ # 解析实现所在的模块 +``` + +```{toctree} + +features.md + +``` diff --git a/docs/flaggems_sglang_zh/reference/operator_list.md b/docs/flaggems_sglang_zh/reference/operator_list.md new file mode 100644 index 0000000000..1aba29691b --- /dev/null +++ b/docs/flaggems_sglang_zh/reference/operator_list.md @@ -0,0 +1,104 @@ +# 算子列表 + +本页列出 FlagGems-sglang 导出的算子,来源于 `conf/operators.yaml` 以及 `src/flaggems_sglang/ops/*.py` 的 `__all__`。 + +通用算子集共 40 个,两个来源的算子名完全一致。每个算子均以 Triton 实现,并通过概览中所述的三级注册器按当前设备解析。 + +## 激活与门控 + +| 算子 | 描述 | +|----------|-------------| +| `gelu_and_mul` | 融合 GELU 门控与乘法:单次遍历计算 gelu(x[..., :d]) * x[..., d:]。具备步长感知能力,非连续的 gate/up 切分无需拷贝即可处理。 | +| `silu_and_mul` | 融合 SiLU 门控与乘法:单次遍历计算 silu(x[..., :d]) * x[..., d:]。针对解码与预填充阶段采用自适应分块策略进行优化。 | +| `silu_and_mul_masked` | 面向分组混合专家激活的带掩码 SiLU 门控与乘法:仅计算各分组掩码长度指示的有效 token 行。 | + +## 注意力 + +| 算子 | 描述 | +|----------|-------------| +| `context_attention` | 打包变长序列上的预填充/上下文阶段缩放点积注意力:支持按批次起始偏移与长度划分的序列上的因果掩码。 | +| `decode_attention` | 针对分页 KV 缓存的解码阶段注意力:通过间接索引表为单 token 查询收集键和值。 | +| `decode_grouped_attention` | 解码阶段分页注意力的分组查询变体:一组查询头共享同一个 KV 头,以降低缓存带宽。 | +| `merge_state` | 使用 log-sum-exp 缩放合并前缀与后缀序列的注意力状态:用于连续批处理与分块注意力计算。 | + +## Mamba 与 SSM + +| 算子 | 描述 | +|----------|-------------| +| `bmm_chunk` | 面向 Mamba2 SSD 扫描的分块批量矩阵乘法:按块计算 C @ B^T 乘积,可选因果掩码。 | +| `causal_conv1d_fn` | 连续批处理序列上的通用深度可分离因果一维卷积:支持可选偏置与 SiLU 激活。 | +| `chunk_state` | Mamba2 分块 SSM 隐状态累加:按块将 x 与经衰减和 dt 缩放的状态投影矩阵累加。 | +| `chunk_state_varlen` | Mamba2 分块状态累加的变长版本:处理由累积序列长度界定的打包序列。 | +| `selective_state_update` | 单步 Mamba 选择性状态空间更新:以 float32 推进循环状态,支持可选 softplus dt、D 跳跃项与门控。 | +| `state_passing` | Mamba2 SSD 块间状态传递扫描:将每个块的最终状态从初始状态起依次向前传播。 | +| `mamba_layernorm_gated` | Mamba 块的门控层归一化/RMS 归一化:在归一化前或后施加分组归一化与 SiLU 门控。 | + +## 分块累积求和 + +| 算子 | 描述 | +|----------|-------------| +| `chunk_cumsum` | Mamba 时间步与状态衰减项的分块累积求和:计算 dt(可选 softplus 与偏置)以及每块的累积 dA。 | +| `chunk_local_cumsum_scalar` | 对分块序列计算局部累积和:支持可选缩放与反向累积方向,兼容直接与转置两种计算方式。 | +| `chunk_local_cumsum_vector` | 逐 token、逐头向量门的块内累积求和:支持可选缩放与反向累积。 | + +## 混合专家(MoE) + +| 算子 | 描述 | +|----------|-------------| +| `fused_moe_gemm` | 融合混合专家 GEMM:将 token 路由与专家计算合并,为 MoE 模型执行按专家加权的矩阵乘法。 | +| `fused_moe_router_cudacore` | 在 CUDA 核心上执行的融合混合专家路由:计算路由 logits(可选 softcapping 与纠正偏置),再选取 top-k 专家。 | +| `fused_moe_router_tensorcore` | 融合混合专家路由的 Tensor Core 变体:与 CUDA 核心路由数学等价,针对 Tensor Core 吞吐进行调优。 | +| `moe_fused_gate` | 带专家分组选择的融合混合专家门控:对专家打分,施加偏置与 top-k 选择,并可选重归一化与重缩放。 | +| `moe_fused_mul_sum` | 融合的混合专家输出加权归约:将每个 top-k 专家输出按其路由权重缩放并对 top-k 求和,屏蔽被专家并行路由丢弃的槽位。 | +| `moe_sum_reduce` | 对混合专家层各专家输出求和:在 top-k 维度上归约并施加路由缩放因子。 | +| `per_group_transpose` | 面向 MoE 模型的按专家分组转置:按专家分配对张量块进行转置,以便高效计算。 | +| `sigmoid_gate_topk_renorm` | 带 top-k 选择与权重重归一化的 sigmoid 专家门控:支持可选偏置、共享专家处理以及路由/全局缩放。 | + +## LoRA + +| 算子 | 描述 | +|----------|-------------| +| `embedding_lora_a` | 面向分段多适配器批次的 LoRA A 投影(作用在 embedding 查表之上):选择各分段的适配器权重并对收集到的行做投影。 | +| `gate_up_lora_b` | 面向分段多适配器批次、用于融合 gate/up 投影的 LoRA B 投影:展开低秩激活并累加到基础输出上。 | +| `qkv_lora_b` | 面向分段多适配器批次、用于融合 QKV 投影的 LoRA B 投影:将低秩激活展开到基础输出的 Q、K、V 各切片偏移上。 | +| `sgemm_lora_a` | 面向分段多适配器批次的 LoRA A 投影 GEMM:将激活收缩到适配器秩,可选择在投影堆叠上执行。 | +| `sgemm_lora_b` | 面向分段多适配器批次的 LoRA B 投影 GEMM:展开低秩激活并累加到基础输出上。 | + +## 归一化 + +| 算子 | 描述 | +|----------|-------------| +| `fused_rmsnorm` | 带可学习权重的融合均方根层归一化:单次遍历计算 RMS 倒数并施加缩放。 | + +## 量化 + +| 算子 | 描述 | +|----------|-------------| +| `per_token_group_quant_int8` | 逐 token、逐分组的对称 INT8 量化:输出量化值以及每个 token 分组一个缩放因子。 | +| `per_token_quant_int8` | 逐 token 对称 INT8 量化:输出量化值以及每个 token 行一个缩放因子。 | + +## 旋转位置编码 + +| 算子 | 描述 | +|----------|-------------| +| `interleaved_rope` | 交错布局的旋转位置编码:按多段的时间、高度、宽度切分旋转相邻元素对。 | +| `mrope_fused` | 面向查询与键张量的多维旋转位置编码(mRoPE):在时间、高度、宽度维度施加位置相关的旋转。 | +| `rotary_embedding` | 使用预计算 cos/sin 表的旋转位置编码:同时支持交错与半切分两种旋转布局。 | + +## 采样 + +| 算子 | 描述 | +|----------|-------------| +| `apply_token_bitmask` | 将打包位掩码施加到 logits 上,用于语法约束解码:把不允许的词表 token 置为 -inf 予以屏蔽。 | +| `softcap_inplace_logits` | 原地 logit softcap:对完整 logit 张量计算 cap * tanh(x / cap),在采样前限制 logit 幅度且不额外分配输出缓冲。 | +| `softcap_out` | 非原地 logit softcap:计算 cap * tanh(x / cap) 并返回 FP32 张量,每次启动使用确定性的形状启发式选择分块。 | + +## 投机解码 + +| 算子 | 描述 | +|----------|-------------| +| `draft_topk1` | 投机解码的 top-1 草稿 token 选择:对每个位置取 argmax 草稿 token 并写入草稿 token 列。 | + +## 多后端算子覆盖 + +除通用实现外,各芯片厂商还在 `src/flaggems_sglang/runtime/backend/_/ops/` 下提供算子特化实现,各厂商的覆盖数量见[特性](../overview/features.md)中的后端表。厂商未特化的算子会自动回退到通用 Triton 实现。 diff --git a/docs/flaggems_sglang_zh/release_notes/release-notes.md b/docs/flaggems_sglang_zh/release_notes/release-notes.md new file mode 100644 index 0000000000..84227b859e --- /dev/null +++ b/docs/flaggems_sglang_zh/release_notes/release-notes.md @@ -0,0 +1,11 @@ +# 发布说明 + +本节包含 FlagGems-sglang 的发布信息。 + +## v0.1.0 + +**新增特性**: + +- 以 Triton 实现了 40 个面向 SGLang 推理的算子,覆盖激活与门控、注意力、Mamba/SSM、分块累积求和、混合专家、LoRA 投影、归一化、INT8 量化、旋转位置编码、采样与投机解码。 +- 灵活的多后端支持:算子经三级注册器(通用 / 厂商 / 架构)解析,已提供昆仑芯、昇腾、燧原、天数智芯、沐曦、海光、摩尔线程与 NVIDIA 的厂商特化实现,并为 NVIDIA 提供 Hopper 与 Ampere 架构目录。 +- 深度性能调优与 Triton 内核调用优化,并配套逐算子的测试与基准套件及精度记录。 diff --git a/docs/flaggems_sglang_zh/user_guide/run-tests-and-benchmark.md b/docs/flaggems_sglang_zh/user_guide/run-tests-and-benchmark.md new file mode 100644 index 0000000000..d8a43a7130 --- /dev/null +++ b/docs/flaggems_sglang_zh/user_guide/run-tests-and-benchmark.md @@ -0,0 +1,40 @@ +# 运行测试与基准 + +本节介绍如何运行 FlagGems-sglang 的测试与基准,以验证正确性并衡量算子性能。 + +以下命令已在 FlagGems-sglang 仓库中验证,可用于安装后的快速检查。 + +## 运行测试 + +```bash +cd FlagGems-sglang +pytest -q tests --collect-only +pytest -q tests/test_silu_and_mul.py --quick +``` + +如需与 CPU 参考实现对比,而不是与设备上的参考实现对比: + +```bash +pytest -q tests/test_silu_and_mul.py --ref cpu --quick +``` + +## 运行基准 + +```bash +cd FlagGems-sglang +pytest -q benchmark --collect-only +pytest -q benchmark/test_silu_and_mul.py --level core --iter 1 --warmup 1 +``` + +基准套件在记录性能的同时也会记录精度。传入 `--record log` 会写出每次运行的日志,传入 `--record json` 则写出 `accuracy_result.json`。 + +```bash +pytest -q benchmark/test_silu_and_mul.py --level core --record log +``` + +```{note} +- 大多数测试/基准需要加速卡运行时;参考实现可通过 `--ref cpu` 在 CPU 上运行。 +- 建议先执行 `--collect-only`,快速确认导入与用例发现是否正常。 +- 如需将运行指向特定厂商后端,请在调用 pytest 前设置 `DNN_VENDOR`。 +- `--level core` 运行核心形状集;默认级别更全面,耗时也更长。 +``` diff --git a/docs/flaggems_sglang_zh/user_guide/usage.md b/docs/flaggems_sglang_zh/user_guide/usage.md new file mode 100644 index 0000000000..e13fd196d0 --- /dev/null +++ b/docs/flaggems_sglang_zh/user_guide/usage.md @@ -0,0 +1,30 @@ +# 使用算子 + +安装 FlagGems-sglang 后,直接导入该包并调用算子即可。针对当前设备解析出的每个算子都挂载在包命名空间上,因此无论底层硬件是什么,调用方式都保持一致。 + +下例使用两个已导出的算子 `silu_and_mul` 与 `fused_rmsnorm`: + +```python +import torch +import flaggems_sglang + +# 创建张量;hidden_states 形状为 [..., 2d] +x = torch.randn(1024, 4096, device=flaggems_sglang.device) + +# 门控 SiLU:out = silu(x[..., :d]) * x[..., d:] +y = flaggems_sglang.silu_and_mul(x) + +# 对最后一维做融合 RMSNorm +weight = torch.ones(2048, device=flaggems_sglang.device) +z = flaggems_sglang.fused_rmsnorm(y, weight, eps=1e-5) +``` + +查看当前机器解析到哪个实现: + +```python +print(flaggems_sglang.device, flaggems_sglang.vendor_name) +print(flaggems_sglang.silu_and_mul.__module__) +print(len(flaggems_sglang.all_registered_ops()), "operators registered") +``` + +完整的算子列表请参见[算子列表](../reference/operator_list.md)。 diff --git a/docs/flaggems_sglang_zh/user_guide/user-guide.md b/docs/flaggems_sglang_zh/user_guide/user-guide.md new file mode 100644 index 0000000000..68b823a3ca --- /dev/null +++ b/docs/flaggems_sglang_zh/user_guide/user-guide.md @@ -0,0 +1,10 @@ +# 用户指南 + +本节指导您如何在 SGLang 推理工作流中使用 FlagGems-sglang 算子,以及如何运行测试与基准。 + +```{toctree} +:maxdepth: 2 + +usage.md +run-tests-and-benchmark.md +``` diff --git a/docs/flagos_homepage/AGENTS.md b/docs/flagos_homepage/AGENTS.md new file mode 100644 index 0000000000..9dde51301d --- /dev/null +++ b/docs/flagos_homepage/AGENTS.md @@ -0,0 +1,24 @@ +# Repository Guidelines + +## Project Structure & Module Organization + +This directory contains the FlagOS documentation homepage, built as one project by the shared `docs/conf.py` Sphinx configuration. Edit `index.md` for the landing page and `overview.md` for the English overview. The `chip_adaptation_guide/` tree contains cloud, edge, and shared adaptation guides; keep new pages in the existing numbered sections and update the relevant `index.md` toctree. Put diagrams and other page assets in `images/` or the closest guide-level `assets/` directory. Homepage styles and behavior belong in `_static/`; Chinese translations are maintained in `locale/zh_CN/LC_MESSAGES/*.po`. + +## Build, Test, and Development Commands + +Run commands from the repository root (`E:\BAAI\github\docs`), after installing `docs/requirements.txt`: + +```powershell +$env:PROJECT="flagos_homepage"; sphinx-build -b html docs docs/_build/html +$env:PROJECT="flagos_homepage"; sphinx-build -b html -D language=zh_CN docs docs/_build/html/zh_CN +``` + +The first command validates the English homepage; the second validates the localized build. For stricter local validation, add `-W --keep-going`. There is no separate unit-test suite for this subtree; a successful Sphinx build and manual browser review are the primary checks. Translation catalogs can be regenerated with `sphinx-build -b gettext` followed by `sphinx-intl update`, and syntax statistics can be checked with `msgfmt --statistics`. + +## Coding Style & Naming Conventions + +Write MyST Markdown using the existing heading hierarchy, directives, and relative links. Use lowercase `snake_case` for new guide filenames (for example, `progress_overview.md`), descriptive alt text for images, and four-space indentation in directive content when needed. Reuse existing `sphinx-design` classes and `_static` CSS patterns before adding new styles. Keep English and Chinese page structure aligned where translations exist. + +## Commit & Pull Request Guidelines + +Use short, imperative commit subjects consistent with history, such as `docs: update homepage links` or `Corrected overview`. Pull requests should explain the affected pages and language, link the relevant issue when applicable, include the exact build command used, and attach screenshots for layout or styling changes. Keep unrelated generated files, backups, and build output out of commits. diff --git a/docs/flagos_homepage/_static/guide.css b/docs/flagos_homepage/_static/guide.css new file mode 100644 index 0000000000..5c8622f058 --- /dev/null +++ b/docs/flagos_homepage/_static/guide.css @@ -0,0 +1,367 @@ +/* FlagOS guide pages visual style. + Scoped away from the portal homepage, whose article contains #documentation. */ + +.bd-page-width:not(:has(#documentation)) { + --flagos-bg: #ffffff; + --flagos-surface: #f7f7f7; + --flagos-surface-raised: #ffffff; + --flagos-border: rgba(14, 15, 17, 0.14); + --flagos-border-soft: rgba(14, 15, 17, 0.08); + --flagos-text: #0e0f11; + --flagos-muted: #646a73; + --flagos-primary: #3752da; + --flagos-primary-soft: rgba(55, 82, 218, 0.1); + --flagos-code-bg: #f5f6f8; + --flagos-radius: 8px; + --pst-font-family-base: "Inter var", Inter, ui-sans-serif, system-ui, -apple-system, BlinkMacSystemFont, "Segoe UI", sans-serif; + --pst-font-family-heading: "Inter var", Inter, ui-sans-serif, system-ui, -apple-system, BlinkMacSystemFont, "Segoe UI", sans-serif; + --pst-font-family-monospace: "SFMono-Regular", Consolas, "Liberation Mono", Menlo, monospace; + --pst-color-primary: var(--flagos-primary); + --pst-color-link: var(--flagos-primary); + --pst-color-link-hover: #233ab6; + --pst-color-background: var(--flagos-bg); + --pst-color-on-background: var(--flagos-surface-raised); + --pst-color-surface: var(--flagos-surface); + --pst-color-border: var(--flagos-border); + --pst-color-text-base: var(--flagos-text); + --pst-color-text-muted: var(--flagos-muted); + --pst-sidebar-primary: 300px; + --pst-sidebar-secondary: 300px; + max-width: none; + background: var(--flagos-bg); + color: var(--flagos-text); +} + +html[data-theme="dark"] .bd-page-width:not(:has(#documentation)) { + --flagos-bg: #101216; + --flagos-surface: #171a20; + --flagos-surface-raised: #13161b; + --flagos-border: rgba(255, 255, 255, 0.14); + --flagos-border-soft: rgba(255, 255, 255, 0.08); + --flagos-text: #eef1f6; + --flagos-muted: #a5adba; + --flagos-primary: #8ea0ff; + --flagos-primary-soft: rgba(142, 160, 255, 0.14); + --flagos-code-bg: #191d24; + --pst-color-primary: var(--flagos-primary); + --pst-color-link: var(--flagos-primary); + --pst-color-link-hover: #b4c0ff; + --pst-color-background: var(--flagos-bg); + --pst-color-on-background: var(--flagos-surface-raised); + --pst-color-surface: var(--flagos-surface); + --pst-color-border: var(--flagos-border); + --pst-color-text-base: var(--flagos-text); + --pst-color-text-muted: var(--flagos-muted); +} + +.bd-page-width:not(:has(#documentation)) .bd-container, +.bd-page-width:not(:has(#documentation)) .bd-main, +.bd-page-width:not(:has(#documentation)) .bd-content { + background: var(--flagos-bg); +} + +@media (min-width: 992px) { + .bd-page-width:not(:has(#documentation)) .bd-sidebar-primary { + flex-basis: 300px; + width: 300px; + max-width: 300px; + } + + input#pst-primary-sidebar-checkbox:checked ~ .bd-container .bd-page-width:not(:has(#documentation)) .bd-sidebar-primary, + .bd-page-width:not(:has(#documentation)) .bd-sidebar-primary.pst-sidebar-hidden { + margin-left: -300px; + } +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-primary { + background: var(--flagos-surface); + border-right: 1px solid var(--flagos-border-soft); + padding: 1rem 0.7rem; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-primary .sidebar-primary-items__start, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-primary .sidebar-primary-items__end { + border: 0; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-primary .sidebar-primary-item { + margin-bottom: 0.25rem; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-primary .toctree-l1 > a, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-primary .toctree-l2 > a, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-primary .toctree-l3 > a, +.bd-page-width:not(:has(#documentation)) .bd-links .nav-link, +.bd-page-width:not(:has(#documentation)) .bd-links .navbar-nav li a { + min-height: 30px; + border-radius: 6px; + color: var(--flagos-muted); + font-size: 0.875rem; + line-height: 1.35; + padding: 0.36rem 0.7rem; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-primary .caption, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-primary p.caption, +.bd-page-width:not(:has(#documentation)) .bd-links p.caption { + margin: 1.05rem 0 0.35rem; + padding: 0 0.7rem; + color: var(--flagos-muted); + font-size: 0.7rem; + font-weight: 700; + letter-spacing: 0; + text-transform: uppercase; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-primary .current > a, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-primary .active > a, +.bd-page-width:not(:has(#documentation)) .bd-links .active > .nav-link, +.bd-page-width:not(:has(#documentation)) .bd-links .nav-link.active { + background: var(--flagos-primary-soft); + color: var(--flagos-primary); + font-weight: 650; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-primary a:hover, +.bd-page-width:not(:has(#documentation)) .bd-links .nav-link:hover { + background: rgba(14, 15, 17, 0.06); + color: var(--flagos-text); + text-decoration: none; +} + +html[data-theme="dark"] .bd-page-width:not(:has(#documentation)) .bd-sidebar-primary a:hover, +html[data-theme="dark"] .bd-page-width:not(:has(#documentation)) .bd-links .nav-link:hover { + background: rgba(255, 255, 255, 0.08); +} + +.bd-page-width:not(:has(#documentation)) .bd-main .bd-content { + flex: 1 1 auto; + max-width: none; + justify-content: flex-start; +} + +.bd-page-width:not(:has(#documentation)) .bd-main .bd-content .bd-article-container { + max-width: 1040px; +} + +.bd-page-width:not(:has(#documentation)) .bd-main .bd-content .bd-article-container .bd-article { + padding: 2rem 2rem 0; +} + +.bd-page-width:not(:has(#documentation)) .bd-article { + color: var(--flagos-text); + font-size: 1rem; + line-height: 1.65; +} + +.bd-page-width:not(:has(#documentation)) .bd-article h1 { + margin: 0 0 1rem; + color: var(--flagos-text); + font-size: 2.15rem; + font-weight: 720; + line-height: 1.15; +} + +.bd-page-width:not(:has(#documentation)) .bd-article h2 { + margin: 2.25rem 0 0.75rem; + color: var(--flagos-text); + font-size: 1.45rem; + font-weight: 680; +} + +.bd-page-width:not(:has(#documentation)) .bd-article h3 { + margin: 1.8rem 0 0.55rem; + color: var(--flagos-text); + font-size: 1.15rem; + font-weight: 680; +} + +.bd-page-width:not(:has(#documentation)) .bd-article p, +.bd-page-width:not(:has(#documentation)) .bd-article li { + color: var(--flagos-text); +} + +.bd-page-width:not(:has(#documentation)) .bd-article p { + margin-bottom: 1rem; +} + +.bd-page-width:not(:has(#documentation)) .bd-article table ol { + margin: 0.2rem 0; + padding-left: 1.35rem; +} + +.bd-page-width:not(:has(#documentation)) .bd-article table ol li { + margin: 0.15rem 0; +} + +.bd-page-width:not(:has(#documentation)) .bd-article a { + text-decoration-thickness: 1px; + text-underline-offset: 0.18em; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary { + width: 300px !important; + min-width: 300px !important; + max-width: 300px !important; + background: var(--flagos-bg); + border-left: 1px solid var(--flagos-border-soft); + padding: 1.1rem 0.8rem 0 0.85rem; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .onthispage { + color: var(--flagos-text); + font-size: 0.8rem; + font-weight: 700; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .sidebar-secondary-items, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .sidebar-secondary__inner, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .sidebar-secondary-item, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary nav.page-toc, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .bd-toc-nav, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .tocsection, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary ul { + box-sizing: border-box; + flex: 1 1 auto; + width: 100%; + max-width: 100%; + min-width: 0; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .sidebar-secondary-item { + border-left: 1px solid var(--flagos-border-soft); + padding-left: 0; + padding-right: 0; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .toc-entry, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .page-toc li { + width: 100%; + max-width: 100%; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .toc-entry a, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .page-toc a, +.bd-page-width:not(:has(#documentation)) .bd-page-toc .toc-entry a { + box-sizing: border-box; + display: block; + width: 100%; + border-left: 2px solid transparent; + color: var(--flagos-muted); + font-size: 0.82rem; + line-height: 1.35; + margin-left: 0; + padding: 0.22rem 0.2rem 0.22rem 0.75rem; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .toc-entry a:hover, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .page-toc a:hover, +.bd-page-width:not(:has(#documentation)) .bd-page-toc .toc-entry a:hover { + color: var(--flagos-text); + text-decoration: none; +} + +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .toc-entry a.active, +.bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .page-toc a.active, +.bd-page-width:not(:has(#documentation)) .bd-page-toc .toc-entry a.active { + border-left-color: var(--flagos-primary); + color: var(--flagos-primary); + font-weight: 650; +} + +.bd-page-width:not(:has(#documentation)) code.literal, +.bd-page-width:not(:has(#documentation)) .bd-article code { + border: 1px solid var(--flagos-border-soft); + border-radius: 5px; + background: var(--flagos-code-bg); + color: var(--flagos-text); + padding: 0.08rem 0.3rem; +} + +.bd-page-width:not(:has(#documentation)) pre, +.bd-page-width:not(:has(#documentation)) div[class*="highlight-"] { + border: 1px solid var(--flagos-border-soft); + border-radius: var(--flagos-radius); + background: var(--flagos-code-bg); +} + +.bd-page-width:not(:has(#documentation)) div[class*="highlight-"] pre { + border: 0; +} + +.bd-page-width:not(:has(#documentation)) .admonition { + border: 1px solid var(--flagos-border); + border-radius: var(--flagos-radius); + box-shadow: none; +} + +.bd-page-width:not(:has(#documentation)) .admonition > .admonition-title { + border-bottom: 1px solid var(--flagos-border-soft); +} + +.bd-page-width:not(:has(#documentation)) .bd-content table { + border: 1px solid var(--flagos-border); + border-radius: var(--flagos-radius); + overflow: hidden; +} + +.bd-page-width:not(:has(#documentation)) .bd-content table thead { + background: var(--flagos-surface); +} + +.bd-page-width:not(:has(#documentation)) .bd-footer-content, +.bd-page-width:not(:has(#documentation)) .bd-footer-article { + border-top-color: var(--flagos-border-soft); +} + +@media (min-width: 1200px) { + .bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary { + position: fixed; + top: calc(var(--pst-header-height, 4rem) + 0.75rem); + left: calc(300px + min(1040px, calc(100vw - 300px - 300px - 3rem)) + 1rem); + right: auto; + height: calc(100vh - var(--pst-header-height, 4rem) - 1.5rem); + overflow-y: auto; + z-index: 10; + } + + .bd-page-width:not(:has(#documentation)):has(.bd-sidebar-primary.hide-on-wide) .bd-main { + margin-left: 300px; + } + + .bd-page-width:not(:has(#documentation)):has(.bd-sidebar-primary.hide-on-wide) .bd-main .bd-content { + max-width: calc(1040px + 300px + 1rem); + } + + .bd-page-width:not(:has(#documentation)) .bd-main .bd-content .bd-article-container { + max-width: 1040px; + } + + .bd-page-width:not(:has(#documentation)) .bd-main .bd-content .bd-article-container .bd-article { + padding-left: 2rem; + padding-right: 2rem; + } +} + +@media (max-width: 1199.98px) { + .bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary { + width: min(92vw, 520px) !important; + min-width: min(92vw, 520px) !important; + max-width: 520px !important; + padding: 2rem 1.25rem 1rem; + } + + .bd-page-width:not(:has(#documentation)) .bd-sidebar-secondary .sidebar-secondary-item { + border-left: 1px solid var(--flagos-border-soft); + } +} + +@media (max-width: 768px) { + .bd-page-width:not(:has(#documentation)) .bd-main .bd-content .bd-article-container .bd-article { + padding: 1rem 1rem 0; + } + + .bd-page-width:not(:has(#documentation)) .bd-article h1 { + font-size: 2rem; + } +} diff --git a/docs/flagos_homepage/_static/homepage.css b/docs/flagos_homepage/_static/homepage.css index 3b8f275479..53bc07f59c 100644 --- a/docs/flagos_homepage/_static/homepage.css +++ b/docs/flagos_homepage/_static/homepage.css @@ -123,6 +123,10 @@ body { box-shadow: 0 4px 12px rgba(59, 130, 246, 0.25); } +.flagos-guide-root-toctree { + display: none; +} + /* Grid layout - 4 columns */ .flagos-grid { display: grid; @@ -1403,4 +1407,4 @@ html[data-theme="dark"] .operator-item ul { html[data-theme="dark"] .operator-item a { color: var(--flagos-blue); -} \ No newline at end of file +} diff --git a/docs/flagos_homepage/chip_adaptation_guide/TODO.md b/docs/flagos_homepage/chip_adaptation_guide/TODO.md new file mode 100644 index 0000000000..f1beacdb5f --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/TODO.md @@ -0,0 +1,25 @@ +# Chip Adaptation Guide Maintenance TODO + +> Progress last updated:2026-09-18 + +## Completed + +- [x] content `chip_adaptation_guide_toctree_backup` +- [x] contentadaptcontent `chip_adaptation_guide` +- [x] content `chip_adaptation_guide` content,content +- [x] content `_shared` content + cloud/edge content stub content +- [x] content `{toctree}`,content `{include}` content +- [x] content `flagos_page_tags.py`,support stub frontmatter `tags` contentContentcontent +- [x] content `_shared` content,content `_shared` content HTML +- [x] content cloud/edge content +- [x] content、contentStage +- [x] content `chip_adaptation_guide` +- [x] content `guide.css` +- [x] Run Docker/Sphinx build validation successfully +- [x] Check generated HTML for the homepage and cloud/edge guides:`guide.css` content,contentpassed `#documentation` content,Contentcontent page TOC content + +## Pending + +- [ ] Delete or archive the old experimental structure `common/`、`variant/`(Currently excluded and does not affect the build; keep it only if comparison is needed) +- [ ] Decide whether to delete the one-time directory migration script `_migrate_chip_experiment.py`(content,contentreference) +- [ ] content git status / diff,contentDescription diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/01_technical_stack/index.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/01_technical_stack/index.md new file mode 100644 index 0000000000..048559c0c7 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/01_technical_stack/index.md @@ -0,0 +1,7 @@ +# FlagOS Technology Stack Overview + +The figure below shows the latest FlagOS technology stack. The FlagOS ecosystem currently includes three categories of open-source projects. The FlagOS open-source core libraries include the FlagGems general-purpose large-model and domain-specific operator libraries, the FlagScale training and inference framework, the FlagTree unified compiler, and the FlagCX unified communication library. We also define ecosystem-enabling projects, including plugins for training and inference engines and open-source tools designed by the FlagOS community, such as KernelGen for automatic operator generation, FlagRelease for automatic migration and release, and FlagPerf for multi-chip evaluation. + +![image.png](../../../images/flagos-architecture-en.png) + +Figure 1. FlagOS technology stack overview diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/02_cooperation_model/cloud.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/02_cooperation_model/cloud.md new file mode 100644 index 0000000000..8b464e17bb --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/02_cooperation_model/cloud.md @@ -0,0 +1,13 @@ +Cooperation is divided into three levels: basic, intermediate, and advanced. Chip companies progress through the levels step by step. + +|**Dimension**|**Basic (integration validation)**|**Intermediate (deep adaptation)**|**Advanced (ecosystem co-building)**| +|---|---|---|---| +|Positioning|Integration validation|Deep adaptation|Ecosystem co-building| +|Operators|Adopt selected high-performance FlagGems operators|Fully support the FlagGems Stable operator set and commit to using FlagGems by default|Co-build FlagGems and generate specialized operators with KernelGen| +|Compiler|Use the FlagTree compiler and C++ Wrapper|Optimize with Hints and TLE; use FlagTree as the default compiler|Co-build all three Triton TLE implementation layers| +|Training/inference framework|Adapt the FlagScale plugin|Model training and inference|Participate in large-model T+0 releases| +|Communication|—|Adapt FlagCX|Enable FlagCX by default and continuously maintain the adaptation code| +|CI/CD|Use the FlagOS CI resource pool for unit tests|Co-build test cases, upstream code, and provide machine resources|Bring in operating-system and cloud partner communities
| +|Compatibility and security|Provide base container image materials and download channels for customized third-party dependencies|Obtain FlagOS certification, promptly update customized packages, and respond immediately to package adaptation issues|Share security working-group results and bring them into downstream distributions| +|Open-source community|Basic member|Intermediate member, eligible to run for PMC|Advanced member, with course and case-study coverage| +|Business expansion|—|Bring in partner clouds and MaaS platforms|Bring in upstream and downstream ecosystem communities| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/02_cooperation_model/edge.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/02_cooperation_model/edge.md new file mode 100644 index 0000000000..154f6be58f --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/02_cooperation_model/edge.md @@ -0,0 +1,11 @@ +Cooperation is divided into three levels: basic, intermediate, and advanced. Chip companies progress through the levels step by step. Given the characteristics of edge chips, no adaptation is required for the FlagCX communication library or the large-model training and inference framework. + +|**Dimension**|**Basic (integration validation)**|**Intermediate (deep adaptation)**|**Advanced (ecosystem co-building)**| +|---|---|---|---| +|Positioning|Integration validation|Deep adaptation|Ecosystem co-building| +|Operators|Adopt selected high-performance FlagGems operators|Fully support the FlagGems Stable operator set and commit to using FlagGems by default|Co-build FlagGems and generate specialized operators with KernelGen
| +|Compiler|Use the FlagTree compiler and C++ Wrapper|Optimize with Hints and TLE; use FlagTree as the default compiler|Co-build all three Triton TLE implementation layers| +|CI/CD|Use the FlagOS CI resource pool for unit tests
|Co-build test cases, upstream code, and provide machine resources. Jointly solve technical issues and complete model adaptation and optimization|Bring in operating-system and cloud partner communities| +|Compatibility and security|Provide base container image materials and download channels for customized third-party dependencies|Obtain FlagOS certification, promptly update customized packages, and respond immediately to package adaptation issues|Share security working-group results and bring them into downstream distributions
| +|Open-source community|Basic member|Intermediate member, eligible to run for PMC|Advanced member, with course and case-study coverage| +|Business expansion|—|Bring in partner clouds and MaaS platforms|Bring in upstream and downstream ecosystem communities| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/02_cooperation_model/index.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/02_cooperation_model/index.md new file mode 100644 index 0000000000..8833d90ccb --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/02_cooperation_model/index.md @@ -0,0 +1 @@ +# Cooperation Level Model diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/cloud.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/cloud.md new file mode 100644 index 0000000000..8b753d534f --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/cloud.md @@ -0,0 +1,12 @@ +Throughout chip adaptation, the chip company must provide physical devices at each stage to support CI/CD, validation, and release. For the first two batches, two machines are recommended for each batch. KernelGen is expected to use two machines so that operators can be specialized, tested in batches, and used for operator competitions. Single-card, four-card, or eight-card configurations should be confirmed according to the scenario. Additional machines are usually required for Day 0, distribution, cloud-native, or showroom plans. Virtualization can supplement multi-OS validation, but critical validation still requires physical devices. + +|**Stage ID**|**Stage**|**Required devices**|**Purpose**|**Configuration requirements**|**Availability timing**| +|---|---|---|---|---|---| +|Stage 1|Environment readiness|1 chip development board or server with remote access for the FlagOS team|Single-card operation|Ubuntu OS, 64 GB or more memory|Prepare immediately after the Stage 0 MOU is signed| +|Stage 2|FlagTree compiler integration|1 device added to the CI resource pool|Compiler build validation and CI pipeline triggers|Single card, 200 GB or more disk space, network access to the FlagOS CI cluster|Connect to CI when Stage 2 starts| +|Stage 3|FlagGems operator adaptation|1 CI machine and 1 debug machine|Automated CI tests and operator development/debugging|The CI machine must stay reliably online; the vendor may keep the debug machine locally|Confirm the CI machine SLA when Stage 3 starts| +|Stage 4a|FlagCX communication adaptation|At least 2 machines, each with at least 4 cards|Collective communication tests and multi-node communication validation|High-speed interconnect such as IB or RoCE; homogeneous cluster|Available when Stage 4a starts| +|Stage 4b|FlagScale training and inference|At least 2 machines, reused from Stage 4a|Distributed training and inference service validation|Same as Stage 4a, with persistent storage mounted|Reuse Stage 4a devices| +|Stage 5|End-to-end validation|1 CI machine and 2 SVT machines|Continuous CI regression and system validation testing|SVT machines must be isolated from CI machines to avoid interference|Available when Stage 5 starts| +|Stage 6|FlagRelease release|1 dedicated FR machine|Model migration, image builds, and release validation|Single card or above, 200 GB or more disk space, high-speed network|Available when Stage 6 starts| +|Stage 7|KernelGen|2 machines|Automatic operator generation, tuning, and operator competitions|Single-card operation with FlagTree compiler and operator development/tuning environments installed|After Stage 2 is complete| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/common_constraints.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/common_constraints.md new file mode 100644 index 0000000000..a132721b44 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/common_constraints.md @@ -0,0 +1,9 @@ +# Key Constraints + +- CI devices must stay online 24/7 and be managed through the unified FlagOS CI/CD resource pool. + +- SVT devices should be physically isolated from CI devices to prevent validation jobs and CI jobs from competing for resources. + +- The dedicated FlagRelease machine should be configured independently and should not be shared with other workloads. + +- All devices must support automated access. Driver and SDK packages should preferably be available through open downloads or long-lived authorization. diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md new file mode 100644 index 0000000000..c777a37eaf --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md @@ -0,0 +1,11 @@ +# Device Investment Evolution + +````{only} flagos_cloud +![image.png](/chip_adaptation_guide/assets/cloud/device_evolution-en.png) +```` + +````{only} flagos_edge +![image.png](/chip_adaptation_guide/assets/edge/device_evolution-en.png) +```` + +Figure 2. Device investment evolution diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/edge.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/edge.md new file mode 100644 index 0000000000..cc767efafb --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/edge.md @@ -0,0 +1,9 @@ +Throughout chip adaptation, the chip company must provide physical devices at each stage to support CI/CD, validation, and release. For the first two batches, two machines are recommended for each batch. A separate machine is required for the FlagRelease stage. Additional machines are usually required for Day 0, distribution, cloud-native, or showroom plans. Virtualization can supplement multi-OS validation, but critical validation still requires physical devices. + +|**Stage ID**|**Stage**|**Required devices**|**Purpose**|**Configuration requirements**|**Availability timing**| +|---|---|---|---|---|---| +|Stage 1|Environment readiness|1 chip development board or server with remote access for the FlagOS team|Single-card operation|Ubuntu OS, 64 GB or more memory|Prepare immediately after the Stage 0 MOU is signed| +|Stage 2|FlagTree compiler integration|1 CI machine added to the CI resource pool and 1 debug machine|Compiler build validation and CI pipeline triggers|Single card, 200 GB or more disk space, network access to the FlagOS CI cluster|Connect to CI when Stage 2 starts| +|Stage 3|FlagGems operator adaptation|2 CI machines and 1 debug machine|Automated CI tests and operator development/debugging|CI machines must stay reliably online; the debug machine should preferably be available to the FlagOS community|Confirm the CI machine SLA when Stage 3 starts| +|Stage 4|End-to-end validation|2 CI machines and 2 SVT machines|Continuous CI regression and system validation testing|SVT machines must be isolated from CI machines to avoid interference|Available when Stage 4 starts| +|Stage 5|FlagRelease release|1 dedicated FR machine|Model migration, image builds, and release validation|Single card or above, 200 GB or more disk space, high-speed network|Available when Stage 5 starts| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/index.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/index.md new file mode 100644 index 0000000000..ec0eee8ad0 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/03_hardware_resources/index.md @@ -0,0 +1 @@ +# Hardware Resource Investment Plan diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/index.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/index.md new file mode 100644 index 0000000000..8907c93193 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/index.md @@ -0,0 +1,9 @@ +# Adaptation Stages and Execution Plan + +````{only} flagos_cloud +The cloud chip path progresses through six stages from kickoff to advanced adaptation. +```` + +````{only} flagos_edge +The edge chip path progresses through the stages from kickoff to advanced adaptation. +```` diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md new file mode 100644 index 0000000000..ccd2f0d70f --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md @@ -0,0 +1,33 @@ +# Overall Progress Overview + +````{only} flagos_cloud +**Overall progress overview:** The figure below shows the complete path from contract kickoff, environment readiness, compiler integration, operator coverage, communication, training and inference, validation, compliance, release, and promotion. It is organized around key deliverables, cooperation-level evolution, and hardware investment. + +**Performance metric requirements:** Starting from Stage 3, operator performance, communication throughput, model accuracy, and end-to-end efficiency gradually become acceptance metrics. For advanced cooperation, performance metrics are a key baseline. Under equivalent compute, the chip must reach a defined percentage of the NVIDIA baseline; the specific percentage will be confirmed later through the KT2 milestone. +```` + +````{only} flagos_edge +**Overall progress overview:** The figure below shows the complete path from contract kickoff, environment readiness, compiler integration, operator coverage, validation, compliance, release, and promotion. It is organized around key deliverables, cooperation-level evolution, and hardware investment. + +**Performance metric requirements:** Starting from Stage 3, operator performance, model accuracy, and end-to-end efficiency gradually become acceptance metrics. For advanced cooperation, performance metrics are a key baseline. +```` + +**Quality and continuous maintenance requirements:** Vendors must commit to maintaining the mainline adaptation code. FlagOS upgrades and vendor Driver/SDK upgrades must both be covered by CI/CD validation so that performance results remain sustainable, reproducible, and releasable. + +````{only} flagos_cloud +![image.png](/chip_adaptation_guide/assets/cloud/progress_overview-en.png) +```` + +````{only} flagos_edge +![image.png](/chip_adaptation_guide/assets/edge/progress_overview-en.png) +```` + +Figure 3. Overall progress overview + +````{only} flagos_cloud +Estimated total duration: about 5-7 months from signing to release. Total hardware investment increases with the cooperation target. The basic path requires about 1 development board, 1 CI machine, 2 SVT machines, and 1 dedicated FR machine. Additional machines should be added for Day 0, distribution, cloud-native, or showroom plans. +```` + +````{only} flagos_edge +Estimated total duration: about 3-5 months from signing to release. Total hardware investment increases with the cooperation target. The basic path requires about 1 development board, 1 CI machine, 2 SVT machines, and 1 dedicated FlagRelease (FR) machine. Additional machines should be added for Day 0, distribution, cloud-native, or showroom plans. +```` diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md new file mode 100644 index 0000000000..95c834a24e --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md @@ -0,0 +1,41 @@ +# Stage 0: Business Kickoff and Contracting + +**Goal** + +Establish the cooperation relationship and clarify the responsibilities of both parties. + +**Prerequisites** + +The chip company has a base SDK/driver and can provide a development environment. + +**Activities** + +1. Sign an MOU defining the cooperation goals, adaptation scope, and division of responsibilities. +2. Connect with the FlagOS cooperation manager and assign contacts for both parties. +3. Set the initial target level for chip adaptation, usually starting at the intermediate level. +4. Sign the Contributor License Agreement (CLA). + +**Deliverables** + +Signed MOU and contact list for both parties. + +**FlagOS contacts** + +Dedicated cooperation manager and technical architect. + +**Chip company contacts** + +Project director and technical lead. + +**Acceptance criteria** + +MOU signed and contacts confirmed by both parties. + +**Estimated duration** + +1-2 weeks + +**Related documents** + +MOU template, CLA, and cooperation handbook + diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md new file mode 100644 index 0000000000..7b37c2fc36 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md @@ -0,0 +1,44 @@ +# Stage 1: Development Board Readiness and Base Environment Adaptation (Chip-Level) + +**Prerequisites** + +The MOU has been signed. + +**Hardware investment** + +1 development board or server, with remote access provided by the vendor to the FlagOS team. + +**Activities** + +1. Provide the chip development board and select at least one operating system, preferably Ubuntu. +2. Provide the driver and SDK access method. Open downloads are preferred. If authorization-only downloads are used, programmatic authentication must support CI/CD automation. +3. Confirm base image compatibility. A neutral image is preferred, followed by a vendor-provided image. +4. Confirm software compatibility, including kernel, Python, and PyTorch community version coverage. +5. Validate the base environment: PyTorch can recognize the target device backend, and the equivalent device discovery API returns an available device. + +**Deliverables** + +Environment readiness report and driver/SDK access documentation. + +**FlagOS contacts** + +Platform engineering team. + +**Chip company contacts** + +Driver and firmware engineers. + +**Acceptance criteria** + +1. The chip can be correctly identified by the OS, for example through lspci or dmidecode. +2. PyTorch can recognize the chip backend. +3. Basic operators, such as matrix multiplication and elementwise operations, execute correctly. +4. The driver can be obtained unattended. + +**Estimated duration** + +1-2 weeks + +**Related documents** + +Environment configuration guide, driver installation manual, and compatibility matrix. diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md new file mode 100644 index 0000000000..f5749a2956 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md @@ -0,0 +1,43 @@ +# Stage 2: Compiler Adaptation - FlagTree Branch Integration (Entry-Level Threshold) + +**Prerequisites** + +The development board environment is ready, and the chip Triton backend or compatibility layer is available. + +**Hardware investment** + +Add the Stage 1 device to the FlagOS CI resource pool and keep it online 24/7. + +**Activities** + +1. Create a protected chip branch in the FlagTree repository. +2. The chip company provides adaptation code or a compatibility layer for the Triton backend. +3. Integrate a C++ Wrapper if a bridge from Triton IR to the vendor SDK is required. +4. Complete instruction mapping through the unified FLIR IR layer. +5. Configure a basic CI pipeline for automatic builds and compilation tests. + +**Deliverables** + +Available FlagTree chip branch and running basic CI pipeline. + +**FlagOS contacts** + +FlagTree compiler team. + +**Chip company contacts** + +Compiler and toolchain engineers. + +**Acceptance criteria** + +1. Triton kernels can be compiled successfully to the chip backend. +2. At least 10 basic operators can be compiled and executed correctly. +3. CI build pass rate is at least 95%, and machine availability is at least 99%. + +**Estimated duration** + +2-4 weeks + +**Related documents** + +[FlagTree integration guide](https://jwolpxeehx.feishu.cn/docx/VIridPa16odD9hxViDLcsD2vnXb): integration plans for GPGPU and DSA/NPU devices.
Triton-TLE: [GitHub documentation](https://github.com/flagos-ai/FlagTree/wiki/TLE), [reference NV implementation](https://github.com/flagos-ai/FlagTree/tree/triton_v3.6.x), [TLE vendor PR template](https://jwolpxeehx.feishu.cn/file/VIOsbkSYZoI3I5xhxdLcuz8snVg?from=from_copylink), [TLE raw CUDA integration vendor example](https://jwolpxeehx.feishu.cn/wiki/Ipvbwa1WYintOwkmNzUchJJun0c).
C++ Wrapper: [libtriton_jit documentation](https://github.com/flagos-ai/libtriton_jit), [sample PR](https://github.com/flagos-ai/libtriton_jit/pull/12/changes), [C++ Runtime multi-backend and Triton operator development sharing](https://jwolpxeehx.feishu.cn/slides/RPkms0jYol2TA7dvbfScjVJWnlu?from=from_copylink). diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md new file mode 100644 index 0000000000..fe6ed29dae --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md @@ -0,0 +1,45 @@ +# Stage 3: Operator Adaptation - FlagGems Coverage (Entry-to-Intermediate Progression) + +**Prerequisites** + +The FlagTree branch is available, and the chip Triton backend is stable. + +**Hardware investment** + +1 CI machine reused from Stage 2 and 1 optional development/debug machine. + +**Activities** + +1. Confirm that the CI machine is connected to the FlagOS CI/CD resource pool. +2. Run the FlagGems test suite and identify failed or incompatible operators. +3. Fix operator compatibility issues, targeting coverage of at least 80% of core operators. +4. Use KernelGen to automatically generate missing operators and validate them on the chip. +5. Tune operator performance, targeting at least 80% of the vendor native operator performance. +6. Establish an SBOM scanning mechanism. + +**Deliverables** + +FlagGems operator compatibility report, test coverage data, and SBOM scanning baseline. + +**FlagOS contacts** + +FlagGems operator team and KernelGen platform team. + +**Chip company contacts** + +Operator development engineers. + +**Acceptance criteria** + +1. FlagGems operator test pass rate is at least 80%. +2. Core LLM operators pass at 100%. +3. Operator performance reaches at least 80% of vendor native performance. +4. The CI test pipeline runs stably for at least two weeks. + +**Estimated duration** + +4-8 weeks + +**Related documents** + +[FlagGems operator library specification (trial)](https://jwolpxeehx.feishu.cn/docx/GawHdXsuRomaQNxpISec30lxnFg)
[KernelGen Wiki](https://docs.flagos.io/projects/kernelgen/en/latest/) diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md new file mode 100644 index 0000000000..d2fc0615d6 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md @@ -0,0 +1,44 @@ +# Stage 5: FlagRelease Release and Commercial Promotion (Intermediate-to-Advanced Progression) + +For edge chip adaptation, this page corresponds to Stage 5. Edge adaptation has no cloud communication or training/inference framework stage, so after Stage 4 End-to-End Validation and Legal Compliance passes, it proceeds directly to FlagRelease release, image availability, and commercial promotion. + +**Stage ID** + +Stage 5 + +**Prerequisites** + +Mainline merge is complete, and certification tests have passed. + +**Hardware investment** + +1 dedicated FlagRelease machine, configured independently and not shared with CI or SVT. + +**Activities** + +1. Obtain official FlagOS certification for the device. +2. Run model migration and image builds on the FlagRelease platform. +3. Release Docker images for this chip, expanding the release list over time. +4. Join the FlagOS large-model T+0 release plan. +5. Integrate with partner cloud and MaaS platforms. +6. Promote the vendor to intermediate or advanced open-source community membership. +7. Prepare joint branding exposure. + +**Deliverables** + +FlagOS certification certificate, FlagRelease image list, and joint branding plan. + +**Acceptance criteria** + +1. FlagOS compatibility certification is passed. +2. At least 10 chip-adapted model images are published on the FlagRelease platform. +3. At least one joint customer case is delivered. +4. Dedicated FlagRelease machine availability is at least 99%. + +**Estimated duration** + +4-6 weeks + +**Related documents** + +Certification standard, image release specification, and joint branding guide
[FlagRelease Wiki](https://docs.flagos.io/projects/FlagRelease/en/latest/) diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md new file mode 100644 index 0000000000..01a77c33be --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md @@ -0,0 +1,43 @@ +# Stage 5: End-to-End Validation and Legal Compliance (Intermediate Closing Gate) + +For cloud chip adaptation, this page corresponds to Stage 5. This stage follows Stage 4a/4b communication and training/inference framework adaptation. It focuses on validating CI/CD, system validation testing, CTS compatibility certification, and legal compliance. + +**Stage ID** + +Stage 5 + +**Prerequisites** + +Technical adaptation is complete for Stages 2-4, and the CI/CD resource pool is ready. + +**Hardware investment** + +1 CI machine and 2 SVT (system validation testing) machines. SVT machines must be physically isolated from CI machines. + +**Activities** + +1. Validate SIG repository functionality. +2. Run DevSecOps automated build, deployment, and validation. +3. Validate SBOM scanning. +4. Run FlagOS CTS compatibility certification tests. +5. Perform legal compliance review. + +**Deliverables** + +Functional validation report, SBOM report, CTS test report, and legal compliance review opinion. + +**Acceptance criteria** + +1. SIG repository test cases pass at 100%. +2. All CTS test items pass or have an explicit waiver list. +3. SBOM scanning reports no critical vulnerabilities. +4. Legal review has no unresolved items. +5. CI/CD and SVT end-to-end automation runs for at least seven days without interruption. + +**Estimated duration** + +3-4 weeks + +**Related documents** + +SIG repository test specification, CTS specification, SBOM standard, compliance review checklist, and DevSecOps practice guide. diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md new file mode 100644 index 0000000000..272fc42e19 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md @@ -0,0 +1,43 @@ +# Stage 4: End-to-End Validation and Legal Compliance (Intermediate Closing Gate) + +For edge chip adaptation, this page corresponds to Stage 4. Edge adaptation has no independent communication framework stage, so it enters End-to-End Validation and Legal Compliance directly after Stage 3. This page reuses the same validation topic but displays Stage 4 for the edge process. + +**Stage ID** + +Stage 4 + +**Prerequisites** + +Technical adaptation is complete for Stages 2-3, and the CI/CD resource pool is ready. + +**Hardware investment** + +2 CI machines and 2 SVT (system validation testing) machines. SVT machines must be physically isolated from CI machines. + +**Activities** + +1. Validate SIG repository functionality. +2. Run DevSecOps automated build, deployment, and validation. +3. Validate SBOM scanning. +4. Run FlagOS CTS compatibility certification tests. +5. Perform legal compliance review. + +**Deliverables** + +Functional validation report, SBOM report, CTS test report, and legal compliance review opinion. + +**Acceptance criteria** + +1. SIG repository test cases pass at 100%. +2. All CTS test items pass or have an explicit waiver list. +3. SBOM scanning reports no critical vulnerabilities. +4. Legal review has no unresolved items. +5. CI/CD and SVT end-to-end automation runs for at least seven days without interruption. + +**Estimated duration** + +3-4 weeks + +**Related documents** + +SIG repository test specification, CTS specification, SBOM standard, compliance review checklist, and DevSecOps practice guide. diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md new file mode 100644 index 0000000000..5f863a2606 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md @@ -0,0 +1,44 @@ +# Stage 6: FlagRelease Release and Commercial Promotion (Intermediate-to-Advanced Progression) + +For cloud chip adaptation, this page corresponds to Stage 6. This stage starts after End-to-End Validation and Legal Compliance are complete. It focuses on FlagOS certification, FlagRelease image release, model migration, and joint commercial promotion. + +**Stage ID** + +Stage 6 + +**Prerequisites** + +Mainline merge is complete, and certification tests have passed. + +**Hardware investment** + +1 dedicated FlagRelease machine, configured independently and not shared with CI or SVT. + +**Activities** + +1. Obtain official FlagOS certification for the device. +2. Run model migration and image builds on the FlagRelease platform. +3. Release Docker images for this chip, expanding the release list over time. +4. Join the FlagOS large-model T+0 release plan. +5. Integrate with partner cloud and MaaS platforms. +6. Promote the vendor to intermediate or advanced open-source community membership. +7. Prepare joint branding exposure. + +**Deliverables** + +FlagOS certification certificate, FlagRelease image list, and joint branding plan. + +**Acceptance criteria** + +1. FlagOS compatibility certification is passed. +2. At least 10 chip-adapted model images are published on the FlagRelease platform. +3. At least one joint customer case is delivered. +4. Dedicated FlagRelease machine availability is at least 99%. + +**Estimated duration** + +4-6 weeks + +**Related documents** + +Certification standard, image release specification, and joint branding guide
[FlagRelease Wiki](https://docs.flagos.io/projects/FlagRelease/en/latest/) diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_architecture.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_architecture.md new file mode 100644 index 0000000000..975aff1b37 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_architecture.md @@ -0,0 +1,6 @@ +# Certification Architecture + +![image5.png](/chip_adaptation_guide/assets/common/certification_architecture-en.png) + +Figure 4. FlagOS certification architecture + diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_levels.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_levels.md new file mode 100644 index 0000000000..5eee6b5795 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_levels.md @@ -0,0 +1,8 @@ +# Certification Levels + +|**Certification level**|**Requirements**|**Applicable objects**|**Grant conditions**| +|---|---|---|---| +|FlagOS Ready|Basic compatibility certification|All chips entering intermediate cooperation|Core CTS test items pass| +|FlagOS Verified|Functionality, performance, and security|Chips that complete intermediate adaptation|CTS, VTS, and STS all pass| +|FlagOS Interoperable|Heterogeneous interoperability certification|Chips entering advanced cooperation|CTS, VTS, STS, and ITS all pass| +|FlagOS Platinum|Flagship certification|T+0 release partners|All tests pass and the partner contributes to the community| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_maintenance.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_maintenance.md new file mode 100644 index 0000000000..56be35ae24 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_maintenance.md @@ -0,0 +1,10 @@ +# Certification Maintenance + +- Validity period: 12 months; recertification is required at expiration. + +- Changes triggering retesting: major driver upgrades, Triton backend refactoring, or major PyTorch upgrades. + +- Certification revocation: a CI pass rate below 80% for three consecutive months, or an unresolved critical security vulnerability. + +- Waiver management: each waiver must state the technical reason, remediation plan, and expected remediation date; the maximum waiver period is six months. + diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_objects.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_objects.md new file mode 100644 index 0000000000..1fdc1a7d9a --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_objects.md @@ -0,0 +1,7 @@ +# Three-Tier Certification Objects + +|**Certification tier**|**Certification object**|**Scope**|**Benefits and listing**| +|---|---|---|---| +|Chip certification|Chip and base software stack|Drivers, SDK, compiler, operators, communication, frameworks, performance, and security|Chip compatibility certification, adaptation list, and Day 0 candidate| +|System certification|Complete server or device|Multi-card interconnect, thermal management, long-duration stability, SVT, resource monitoring, and system stability|System certification and showroom candidate| +|Software/cloud-native certification|OS distributions, containers, Kubernetes, and cloud-native platforms|Images, scheduling, Kubernetes, system integration, security, and interoperability|Distribution plan and cloud-native ecosystem certification| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_process.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_process.md new file mode 100644 index 0000000000..c4688bde74 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/certification_process.md @@ -0,0 +1,6 @@ +# Certification Process + +![image6.png](/chip_adaptation_guide/assets/common/certification_process-en.jpeg) + +Figure 5. Certification process + diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/index.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/index.md new file mode 100644 index 0000000000..6c53a94d2a --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/index.md @@ -0,0 +1,3 @@ +# FlagOS Compatibility Certification System + +Following the Android CDD/CTS design, FlagOS has established a tiered certification system. Based on further discussion, certification objects are divided into three layers: chips, complete systems, and software/cloud-native platforms, covering different vendor forms and later showroom, policy, and procurement scenarios. diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_cloud.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_cloud.md new file mode 100644 index 0000000000..c48d564d1a --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_cloud.md @@ -0,0 +1,34 @@ +# Test Suites + +## CTS (Compatibility Test Suite) - Compatibility Test Suite + +|**Test domain**|**Coverage**|**Pass standard**| +|---|---|---| +|Compiler compatibility|FlagTree compilation and FLIR IR mapping correctness|100%| +|Operator compatibility|FlagGems core operator correctness|>=95%| +|Communication compatibility|FlagCX collective communication correctness|100%| +|Framework compatibility|Basic FlagScale training and inference workflow|100%| +|PyTorch API compatibility|Correctness of common aten API return values|>=90%| +|Triton API compatibility|Triton language feature support|>=95%| +|Accuracy consistency|Comparison with the NVIDIA baseline|Deviation <2%| + +```{include} ../../_shared/05_compatibility_certification/test_suites_common.md +``` + +## STS (Security Test Suite) - Security Test Suite + +|**Test domain**|**Coverage**|**Pass standard**| +|---|---|---| +|SBOM scanning|Dependency vulnerability scanning|No high-risk vulnerabilities or CVEs| +|Driver security|Driver file permission and signature validation|Passed| +|Container security|Container image scanning|No high-risk vulnerabilities| +|Communication encryption|FlagCX communication-link encryption validation|Passed| + +## ITS (Interoperability Test Suite) - Interoperability Test Suite + +|**Test domain**|**Coverage**|**Pass standard**| +|---|---|---| +|Heterogeneous mixed training|Chip and NVIDIA mixed training|Efficiency >=81%| +|Heterogeneous mixed inference|Chip and NVIDIA mixed inference|Accuracy deviation <2%| +|Cross-vendor communication|FlagCX cross-vendor testing|Correctness 100%| +|Multi-framework compatibility|FlagScale, vLLM, and SGLang|All passed| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_common.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_common.md new file mode 100644 index 0000000000..408750c2d9 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_common.md @@ -0,0 +1,8 @@ +## VTS (Verification Test Suite) - Verification Test Suite + +|**Test domain**|**Coverage**|**Pass standard**| +|---|---|---| +|Stress test|Continuous operator execution for 72 hours|No crashes| +|Stability test|1,000 repeated training and inference runs|Consistent results| +|Long-duration stability|Continuous 24/7 operation|No memory leaks| +|Resource monitoring|GPU utilization, memory bandwidth, and temperature|Within vendor specifications| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_edge.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_edge.md new file mode 100644 index 0000000000..ee27ceb4ec --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_edge.md @@ -0,0 +1,34 @@ +# Test Suites + +## CTS (Compatibility Test Suite) - Compatibility Test Suite + +|**Test domain**|**Coverage**|**Pass standard**| +|---|---|---| +|Compiler compatibility|FlagTree compilation and FLIR IR mapping correctness|100%| +|Operator compatibility|FlagGems core operator correctness|>=95%| +|Communication compatibility (not required for edge chips)|FlagCX collective communication correctness|100%| +|Framework compatibility (not required for edge chips)|Basic FlagScale training and inference workflow|100%| +|PyTorch API compatibility|Correctness of common aten API return values|>=90%| +|Triton API compatibility|Triton language feature support|>=95%| +|Accuracy consistency|Comparison with the NVIDIA baseline|Deviation <2%| + +```{include} ../../_shared/05_compatibility_certification/test_suites_common.md +``` + +## STS (Security Test Suite) - Security Test Suite + +|**Test domain**|**Coverage**|**Pass standard**| +|---|---|---| +|SBOM scanning|Dependency vulnerability scanning|No high-risk vulnerabilities or CVEs| +|Driver security|Driver file permission and signature validation|Passed| +|Container security|Container image scanning|No high-risk vulnerabilities| +|Communication encryption (not required for edge chips)|FlagCX communication-link encryption validation|Passed| + +## ITS (Interoperability Test Suite) - Interoperability Test Suite + +|**Test domain**|**Coverage**|**Pass standard**| +|---|---|---| +|Heterogeneous mixed training|Chip and NVIDIA mixed training|Efficiency >=81%| +|Heterogeneous mixed inference|Chip and NVIDIA mixed inference|Accuracy deviation <2%| +|Cross-vendor communication (not required for edge chips)|FlagCX cross-vendor testing|Correctness 100%| +|Multi-framework compatibility|FlagScale, vLLM, and SGLang|All passed| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md new file mode 100644 index 0000000000..163857835c --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md @@ -0,0 +1,26 @@ +# AI Distribution Validation + +**Validation party** + +AI distribution company + +**Validation scope** + +1. Complete installation and operation of PyTorch on the chip. +2. Loading and execution of the FlagGems operator library. +3. End-to-end operation of the Triton compilation pipeline. +4. Loading and inference of representative models such as Qwen2.5 and Llama3. + +**Hardware requirements** + +The dedicated machine provided by the chip company during the FlagRelease stage can be made available for remote use by the AI distribution. + +**Acceptance criteria** + +1. PyTorch core test cases pass. +2. The FlagGems operator loading rate reaches 100%. +3. Triton kernels compile and execute correctly. + +**Relationship to certification** + +AI distribution validation is an important reference condition for FlagOS Platinum certification. diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/compatibility_declaration.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/compatibility_declaration.md new file mode 100644 index 0000000000..377ea13d4b --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/compatibility_declaration.md @@ -0,0 +1,8 @@ +# Compatibility Declaration + +|**Declaration level**|**Meaning**|**Condition**| +|---|---|---| +|Platinum Partner|The chip is fully supported in the distribution.|Passes Level 1 + Level 2 + Level 3 validation and is an advanced partner.| +|Certified|The chip is certified to run normally in the distribution.|Passes Level 1 + Level 2 validation.| +|Compatible|The chip is verified as compatible in the distribution.|Passes at least Level 1 validation.| +|Community Support|Community-driven support without official distribution validation.|Only FlagOS CTS has passed.| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/index.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/index.md new file mode 100644 index 0000000000..a6b02fd9ad --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/index.md @@ -0,0 +1,3 @@ +# Distribution Validation System (Draft) + +End users obtain FlagOS capabilities through operating-system and AI distributions. Distribution companies must validate chip compatibility in their distributions so that adaptation results move beyond experimental environments into real delivery pipelines. diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/joint_validation.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/joint_validation.md new file mode 100644 index 0000000000..c5b26dd4a8 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/joint_validation.md @@ -0,0 +1,10 @@ +# Joint Validation Process + +|**Step**|**Action**|**Result**| +|---|---|---| +|1|The chip passes FlagOS CTS|CTS report and certification level| +|2|FlagOS sends the chip certification information to the distribution company|Certification level, CTS report, and device access method| +|3|OS distribution validation|OS distribution validation report| +|4|AI distribution validation|AI distribution validation report| +|5|ISV validation (optional)|Third-party compatibility declaration| +|6|The distribution lists the chip|Entry in the "FlagOS Compatible" list| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md new file mode 100644 index 0000000000..1e25328378 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md @@ -0,0 +1,26 @@ +# OS Distribution Validation + +**Validation party** + +OS distribution company + +**Validation scope** + +1. Driver loading and stability on the target OS kernel version. +2. Compatibility of core system libraries such as glibc, libstdc++, and OpenMPI. +3. Docker/containerd runtime operation. +4. Kernel module signature validation. + +**Hardware requirements** + +One device provided by the chip company or made available for remote access. + +**Acceptance criteria** + +1. The driver loads normally on the target kernel, with no errors in dmesg. +2. Core LTP test items pass. +3. The container runtime can pull and run standard images. + +**Relationship to certification** + +After the chip obtains FlagOS Verified certification, it enters the OS distribution validation pipeline. diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/validation_levels.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/validation_levels.md new file mode 100644 index 0000000000..702f03c749 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/06_distribution_validation/validation_levels.md @@ -0,0 +1,6 @@ +# Distribution Validation Levels + +![image7.png](/chip_adaptation_guide/assets/common/distribution_validation-en.png) + +Figure 6. Distribution validation levels + diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/07_quality_metrics/cloud.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/07_quality_metrics/cloud.md new file mode 100644 index 0000000000..01e1d6b883 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/07_quality_metrics/cloud.md @@ -0,0 +1,13 @@ +**Performance-related metrics must remain key acceptance criteria throughout Stages 3-6.** Stage 3 focuses on operator performance, Stage 4a on communication throughput and linearity, and Stage 4b on training/inference accuracy and end-to-end performance. Before advanced cooperation, Day 0, distribution, or showroom plans, the efficiency target relative to the NVIDIA baseline under equivalent compute must be confirmed. The exact percentage will be defined later according to the KT2 requirements. + +|**Stage**|**Acceptance criteria**|**Pass standard**|**Required hardware**| +|---|---|---|---| +|Stage 0|MOU signing|Effective after both parties sign or seal|N/A| +|Stage 1|Environment readiness|Chip is recognized by the OS and PyTorch can recognize the backend|1 development board| +|Stage 2|Triton kernel compilation pass rate|>=95%, CI machine availability >=99%|1 CI machine| +|Stage 3|FlagGems operator test pass rate|>=80%, core operators 100%|1 CI machine and 1 debug machine| +|Stage 4a|Homogeneous collective communication throughput ratio|>=99% of the native library|2-machine multi-card cluster| +|Stage 4b|Model accuracy deviation|<2% relative to the NVIDIA baseline|Reuse Stage 4a devices| +|Stage 5|CTS core tests passed|100% passed or waiver approved|1 CI machine and 2 SVT machines| +|Stage 6|Number of models released through FlagRelease|>=10 models|1 dedicated FR machine| +|Advanced cooperation|Equivalent-compute performance comparison|Efficiency reaches X% of the NVIDIA equivalent-compute baseline|Multi-card cluster, FR machine, or CI| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/07_quality_metrics/edge.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/07_quality_metrics/edge.md new file mode 100644 index 0000000000..2baf2bb877 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/07_quality_metrics/edge.md @@ -0,0 +1,11 @@ +**Performance-related metrics must remain key acceptance criteria throughout Stages 3-6.** Stage 3 focuses on operator performance, and Stage 4 focuses on inference accuracy and end-to-end performance. + +|**Stage**|**Acceptance criteria**|**Pass standard**|**Required hardware**| +|---|---|---|---| +|Stage 0|MOU signing|Effective after both parties sign or seal|N/A| +|Stage 1|Environment readiness|Chip is recognized by the OS and PyTorch can recognize the backend|1 development board| +|Stage 2|Triton kernel compilation pass rate|>=95%, CI machine availability >=99%|1 CI machine| +|Stage 3|FlagGems operator test pass rate|>=80%, core operators 100%|2 CI machines and 1 debug machine| +|Stage 4|Model accuracy deviation|<2% relative to the NVIDIA baseline|2 CI machines and 1 debug machine| +|Stage 5|CTS core tests passed|100% passed or waiver approved|2 CI machines and 2 SVT machines| +|Stage 6|Number of models released through FlagRelease|>=10 models|1 dedicated FR machine| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/07_quality_metrics/index.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/07_quality_metrics/index.md new file mode 100644 index 0000000000..15da2fbfcc --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/07_quality_metrics/index.md @@ -0,0 +1 @@ +# Key Quality Metrics diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/08_tools_and_skills/skill_catalog.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/08_tools_and_skills/skill_catalog.md new file mode 100644 index 0000000000..43e5f9c3af --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/08_tools_and_skills/skill_catalog.md @@ -0,0 +1,13 @@ +# Supporting Skills and Automation Tools + +FlagOS packages multiple adaptation workflow steps as reusable Agent Skills. Chip companies and their development teams can use them according to the adaptation stage. + +|**Skill name**|**Applicable stage**|**Function**| +|---|---|---| +|gpu-container-setup-flagos|Stage 1|Automatically detect the GPU vendor, launch a PyTorch container, and validate the GPU/accelerator card.| +|install-stack-flagos|Stages 1-2|Install the full stack: FlagTree, FlagGems, FlagCX, and FlagScale.| +|kernelgen-flagos|Stage 3|Automatically generate and iteratively optimize operators.| +|model-migrate-flagos|Stage 4b|Backport models from upstream vLLM to vllm-plugin-FL.| +|model-verify-flagos|Stage 5|Verify the serving stack step by step.| +|perf-test-flagos|Stages 4-5|Run performance benchmarks.| +|flagrelease-entrance-flagos|Stage 6|Orchestrate the complete release pipeline.| diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/09_promotion_criteria/advanced_quality_requirements.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/09_promotion_criteria/advanced_quality_requirements.md new file mode 100644 index 0000000000..fc95517d25 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/09_promotion_criteria/advanced_quality_requirements.md @@ -0,0 +1,9 @@ +# Advanced Cooperation: Deterministic Performance and Quality Requirements + +- Performance metrics are prerequisites for advanced cooperation, Day 0 plans, distribution validation, and showroom display; they must not be treated as post-release promotional items. + +- Vendors must maintain the mainstream diff so that chip adaptation on the mainline has continuous ownership. + +- FlagOS upgrades and vendor Driver/SDK upgrades must both enter CI/CD validation. + +- Day 0 releases, distribution plans, and policy recommendation plans must meet maturity, support, performance, and quality requirements. diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/09_promotion_criteria/index.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/09_promotion_criteria/index.md new file mode 100644 index 0000000000..f99890de58 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/09_promotion_criteria/index.md @@ -0,0 +1 @@ +# Promotion Criteria Checklist diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md new file mode 100644 index 0000000000..8c1c57669f --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md @@ -0,0 +1,17 @@ +# Promotion from Intermediate to Advanced + +- If the hardware supports FlagCX, complete the adaptation and maintain it continuously; the SVT two-machine cluster must remain stable. + +- Use the FlagTree compiler by default. + +- Pass the FlagOS CTS, VTS, and STS certification tests. + +- Establish a compatibility matrix for core dependencies such as PyTorch and Triton. + +- Complete the DevSecOps system and close the loop for security incident response. + +- Co-build the FlagGems operator library and contribute code. + +- Prepare the dedicated FlagRelease machine and continuously publish model images. + +- Work with operating-system distributions and cloud partner communities. diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md new file mode 100644 index 0000000000..3b01176f6c --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md @@ -0,0 +1,17 @@ +# Requirements for promotion from basic to intermediate + +- The FlagTree chip branch runs stably for at least four weeks + +- FlagGems operator test coverage is at least 80% + +- The CI single machine is in the FlagOS resource pool with availability of at least 99% + +- The chip passes Sig repository functional validation. + +- Establish an SBOM scanning mechanism + +- The driver/SDK can be obtained unattended + +- Sign and pass legal compliance review + + diff --git a/docs/flagos_homepage/chip_adaptation_guide/_shared/10_risks_and_maintenance/index.md b/docs/flagos_homepage/chip_adaptation_guide/_shared/10_risks_and_maintenance/index.md new file mode 100644 index 0000000000..448426792b --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/_shared/10_risks_and_maintenance/index.md @@ -0,0 +1,11 @@ +# Risks and Maintenance + +|**Risk or requirement**|**Description**| +|---|---| +|Driver access method|Refreshing token authentication every 30 minutes is impractical for automation. Driver and SDK packages should be openly downloadable or use long-lived authorization.| +|Specific dependency trade-offs|Some trade-offs are allowed at the intermediate stage, but specific dependencies must be eliminated before advanced cooperation.| +|Image build neutrality|Use neutral base images to demonstrate compatibility rather than relying entirely on vendor-customized images.| +|Long-term maintenance commitment|Advanced cooperation requires continuous security maintenance and progressive improvement of TLS and other version lifecycle management.| +|Heterogeneous mixed efficiency|When a chip is used in a heterogeneous cluster at the intermediate stage, pay particular attention to mixed communication efficiency of at least 81%.| +|Device SLA|CI devices must stay online 24/7, SVT machines must be isolated, and the FR machine must be configured independently. Long-term offline status may cause certification downgrade or revocation.| +|Waiver period|Each certification-test waiver may last no more than six months. An unresolved waiver must be resubmitted or may result in certification downgrade.| diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/cloud/device_evolution-en.png b/docs/flagos_homepage/chip_adaptation_guide/assets/cloud/device_evolution-en.png new file mode 100644 index 0000000000..081c55c0f7 Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/cloud/device_evolution-en.png differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/cloud/device_evolution.png b/docs/flagos_homepage/chip_adaptation_guide/assets/cloud/device_evolution.png new file mode 100644 index 0000000000..b839a08940 Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/cloud/device_evolution.png differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/cloud/progress_overview-en.png b/docs/flagos_homepage/chip_adaptation_guide/assets/cloud/progress_overview-en.png new file mode 100644 index 0000000000..945232eb04 Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/cloud/progress_overview-en.png differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/cloud/progress_overview.png b/docs/flagos_homepage/chip_adaptation_guide/assets/cloud/progress_overview.png new file mode 100644 index 0000000000..0aff95c311 Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/cloud/progress_overview.png differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/common/certification_architecture-en.png b/docs/flagos_homepage/chip_adaptation_guide/assets/common/certification_architecture-en.png new file mode 100644 index 0000000000..d22e282a14 Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/common/certification_architecture-en.png differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/common/certification_architecture.png b/docs/flagos_homepage/chip_adaptation_guide/assets/common/certification_architecture.png new file mode 100644 index 0000000000..316b7a4bbb Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/common/certification_architecture.png differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/common/certification_process-en.jpeg b/docs/flagos_homepage/chip_adaptation_guide/assets/common/certification_process-en.jpeg new file mode 100644 index 0000000000..3a9b54ea95 Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/common/certification_process-en.jpeg differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/common/certification_process.png b/docs/flagos_homepage/chip_adaptation_guide/assets/common/certification_process.png new file mode 100644 index 0000000000..291d7ae2fc Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/common/certification_process.png differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/common/distribution_validation-en.png b/docs/flagos_homepage/chip_adaptation_guide/assets/common/distribution_validation-en.png new file mode 100644 index 0000000000..7091a33aff Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/common/distribution_validation-en.png differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/common/distribution_validation.png b/docs/flagos_homepage/chip_adaptation_guide/assets/common/distribution_validation.png new file mode 100644 index 0000000000..be8ed09650 Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/common/distribution_validation.png differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/edge/device_evolution-en.png b/docs/flagos_homepage/chip_adaptation_guide/assets/edge/device_evolution-en.png new file mode 100644 index 0000000000..56d001bcdf Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/edge/device_evolution-en.png differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/edge/device_evolution.png b/docs/flagos_homepage/chip_adaptation_guide/assets/edge/device_evolution.png new file mode 100644 index 0000000000..83a5da7a2a Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/edge/device_evolution.png differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/edge/progress_overview-en.png b/docs/flagos_homepage/chip_adaptation_guide/assets/edge/progress_overview-en.png new file mode 100644 index 0000000000..c067986681 Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/edge/progress_overview-en.png differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/assets/edge/progress_overview.png b/docs/flagos_homepage/chip_adaptation_guide/assets/edge/progress_overview.png new file mode 100644 index 0000000000..c78c7e23d5 Binary files /dev/null and b/docs/flagos_homepage/chip_adaptation_guide/assets/edge/progress_overview.png differ diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/01_technical_stack/index.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/01_technical_stack/index.md new file mode 100644 index 0000000000..e93c811754 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/01_technical_stack/index.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/01_technical_stack/index.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/02_cooperation_model/index.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/02_cooperation_model/index.md new file mode 100644 index 0000000000..8689e11c39 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/02_cooperation_model/index.md @@ -0,0 +1,9 @@ +--- +tags: [flagos_cloud] +--- + +```{include} ../../_shared/02_cooperation_model/index.md +``` + +```{include} ../../_shared/02_cooperation_model/cloud.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/03_hardware_resources/common_constraints.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/03_hardware_resources/common_constraints.md new file mode 100644 index 0000000000..c70df1560b --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/03_hardware_resources/common_constraints.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/03_hardware_resources/common_constraints.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/03_hardware_resources/device_evolution.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/03_hardware_resources/device_evolution.md new file mode 100644 index 0000000000..12d6657565 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/03_hardware_resources/device_evolution.md @@ -0,0 +1,6 @@ +--- +tags: [flagos_cloud] +--- + +```{include} ../../_shared/03_hardware_resources/device_evolution.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/03_hardware_resources/index.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/03_hardware_resources/index.md new file mode 100644 index 0000000000..1d7c6a2b13 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/03_hardware_resources/index.md @@ -0,0 +1,16 @@ +--- +tags: [flagos_cloud] +--- + +```{include} ../../_shared/03_hardware_resources/index.md +``` + +```{include} ../../_shared/03_hardware_resources/cloud.md +``` + +```{toctree} +:maxdepth: 1 + +device_evolution.md +common_constraints.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/index.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/index.md new file mode 100644 index 0000000000..02492d0ba3 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/index.md @@ -0,0 +1,21 @@ +--- +tags: [flagos_cloud] +--- + +```{include} ../../_shared/04_adaptation_plan/index.md +``` + +```{toctree} +:maxdepth: 1 + +progress_overview.md +stage_00_business.md +stage_01_environment.md +stage_02_flagtree.md +stage_03_flaggems.md +stage_04_communication_framework.md +stage_05_validation.md +stage_06_release.md +stage_07_kernelgen.md +stage_08_image_build.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/progress_overview.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/progress_overview.md new file mode 100644 index 0000000000..fde8c4b348 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/progress_overview.md @@ -0,0 +1,6 @@ +--- +tags: [flagos_cloud] +--- + +```{include} ../../_shared/04_adaptation_plan/progress_overview.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_00_business.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_00_business.md new file mode 100644 index 0000000000..26e651621d --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_00_business.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/04_adaptation_plan/stage_00_business.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_01_environment.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_01_environment.md new file mode 100644 index 0000000000..dffbea5268 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_01_environment.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/04_adaptation_plan/stage_01_environment.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_02_flagtree.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_02_flagtree.md new file mode 100644 index 0000000000..d73e54383c --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_02_flagtree.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/04_adaptation_plan/stage_02_flagtree.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_03_flaggems.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_03_flaggems.md new file mode 100644 index 0000000000..1a046aecf8 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_03_flaggems.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/04_adaptation_plan/stage_03_flaggems.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md new file mode 100644 index 0000000000..880f4eb221 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md @@ -0,0 +1,89 @@ +# Stage 4: Communication and Training/Inference Framework Adaptation (Intermediate Core) + +## Sub-stage 4a: Communication Adaptation - FlagCX + +**Prerequisites** + +Operator coverage reaches the intermediate standard. + +**Hardware investment** + +At least two servers, each with at least four cards, connected through high-speed interconnect such as IB or RoCE to form a homogeneous cluster. + +**Activities** + +1. Wrap the vendor native communication library as a unified FlagCX adapter. +2. Implement core collective communication operations: allreduce, reducescatter, allgather, send, and recv. +3. Validate homogeneous cluster communication performance, targeting at least 99% of the native library. +4. Validate heterogeneous cluster mixed communication efficiency, targeting at least 81%. + +**Deliverables** + +FlagCX communication adapter and communication performance test report. + +**FlagOS contacts** + +FlagOS framework R&D team. + +**Chip company contacts** + +Communication library and network engineers. + +**Acceptance criteria** + +1. Collective communication executes correctly in single-node multi-card environments. +2. Linearity reaches at least 90% on clusters with eight or more nodes. +3. Homogeneous cluster throughput reaches at least 99% of the native library. +4. Heterogeneous mixed communication efficiency reaches at least 81%. + +**Estimated duration** + +3-5 weeks + +**Related documents** + +[FlagCX vendor adaptation guide](https://jwolpxeehx.feishu.cn/docx/Vl1NdA8jiohjl6xZV2GcfV6un4d) + +## Sub-stage 4b: Training and Inference Adaptation - FlagScale + +**Prerequisites** + +FlagCX communication validation has passed. + +**Hardware investment** + +Reuse the two Stage 4a devices, with persistent storage mounted. + +**Activities** + +1. Adapt the Megatron-LM plugin for the training backend. +2. Adapt the vLLM/SGLang plugins for the inference backend. +3. Support TP, PP, DP, EP, and other parallel strategies. +4. Support FP16, BF16, and FP8 mixed-precision training. +5. Adapt vllm-plugin-FL and provide multi-chip inference services. + +**Deliverables** + +FlagScale integration report and training/inference validation results. + +**FlagOS contacts** + +FlagOS framework R&D team. + +**Chip company contacts** + +Framework engineers. + +**Acceptance criteria** + +1. A representative model can complete one iteration. +2. The inference service responds normally end to end, with accuracy deviation below 2%. +3. The distributed training convergence curve matches the NVIDIA baseline. + +**Estimated duration** + +3-5 weeks + +**Related documents** + +[FlagOS reference material](https://jwolpxeehx.feishu.cn/docx/IiAXdz9ZWouEi0xqMMCcUUU7nJe)
[TransformerEngine-FL plugin mechanism user guide](https://jwolpxeehx.feishu.cn/docx/HC0mdLIOFoGv0sxyo8XcFloinOd)
[vLLM-plugin-FL plugin mechanism user guide](https://jwolpxeehx.feishu.cn/docx/Oxeyd2WpMo3XTCxqRNMcSCKHngd)
[sglang-plugin-FL vendor adaptation development guide](https://jwolpxeehx.feishu.cn/wiki/ReW9w8Kguihzm3kAu8GcSIjsn34)
[veRL-FL adaptation documentation](https://jwolpxeehx.feishu.cn/docx/WF7Kdd8ploywmlxBETfc0CWonne) diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_05_validation.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_05_validation.md new file mode 100644 index 0000000000..d22b440366 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_05_validation.md @@ -0,0 +1,6 @@ +--- +tags: [flagos_cloud] +--- + +```{include} ../../_shared/04_adaptation_plan/stage_05_validation_cloud.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_06_release.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_06_release.md new file mode 100644 index 0000000000..844c42da40 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_06_release.md @@ -0,0 +1,6 @@ +--- +tags: [flagos_cloud] +--- + +```{include} ../../_shared/04_adaptation_plan/stage_06_release_cloud.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md new file mode 100644 index 0000000000..e94fc45c3c --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md @@ -0,0 +1,45 @@ +# Stage 7: KernelGen Adaptation, Automatic Operator Generation, and Operator Competition + +**Prerequisites** + +Mainline merge is complete, FlagOS certification has passed, the FlagRelease image has been released, and the chip compiler, runtime, profiler, and base operator library are available. + +**Hardware investment** + +At least one dedicated KernelGen machine. Multi-card test environments are required for multi-card or multi-node optimization. Operator competitions require a separate evaluation machine or isolated queue. + +**Activities** + +1. Connect to the KernelGen automatic operator generation platform. +2. Configure compilation, runtime, validation, and benchmark environments for the chip. +3. Connect the chip profiler for automatic performance bottleneck analysis. +4. Establish performance baselines for PyTorch, Triton, TLE, and native libraries. +5. Prioritize high-frequency LLM operators, including Attention, MoE, RMSNorm, RoPE, TopK, KV Cache, communication fusion, and others. +6. Automatically generate operators and submit PRs to FlagGems and FlagOS. +7. Build the chip's KernelGen Skills and optimization knowledge base. +8. Connect to the KernelGenBench multi-chip evaluation leaderboard. +9. Co-host an operator optimization competition or KernelGen Hackathon. +10. Include outstanding operators in the FlagRelease image and large-model T+0 release process. + +**Deliverables** + +KernelGen chip adaptation report, operator generation environment, profiler integration documentation, performance baseline report, generated-operator PR list, merged-operator list, KernelGen Skills, KernelGenBench results, operator competition plan, and leaderboard. + +**Acceptance criteria** + +1. Complete the automatic generation, compilation, correctness validation, benchmark, and PR loop. +2. Cover at least 30 high-frequency operators in the first batch. +3. Produce at least 10 mergeable PRs. +4. Merge at least 5 FlagGems or FlagOS operators. +5. Achieve at least 1.1x average speedup for key operators with no performance regression. +6. Document at least 20 chip optimization Skills. +7. Cover at least one official KernelGenBench track. +8. Deliver at least one joint technical case study. + +**Estimated duration** + +6-10 weeks + +**Related documents** + +KernelGen chip integration specification, operator generation specification, FlagGems PR specification, KernelGenBench evaluation specification, profiler integration guide, and operator competition rules. diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.md new file mode 100644 index 0000000000..b8a4a902fe --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.md @@ -0,0 +1,10 @@ +# Stage 8: Image Build + +Image builds are divided into three levels: base image, runtime image, and application image. + +Follow these documentation guides: + +- [Release Info Overview](https://flagos-ai.github.io/release-info/overview/) +- [Release Info Onboarding Guide](https://flagos-ai.github.io/release-info/contribution/onboarding/) + +If you would like to join, submit an issue in the [release-info repository](https://github.com/flagos-ai/release-info/issues). diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_architecture.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_architecture.md new file mode 100644 index 0000000000..5609928de8 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_architecture.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/05_compatibility_certification/certification_architecture.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md new file mode 100644 index 0000000000..03fe141aca --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/05_compatibility_certification/certification_levels.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_maintenance.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_maintenance.md new file mode 100644 index 0000000000..d55723d1f7 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_maintenance.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/05_compatibility_certification/certification_maintenance.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md new file mode 100644 index 0000000000..84d2927749 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/05_compatibility_certification/certification_objects.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_process.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_process.md new file mode 100644 index 0000000000..f4bad71ad0 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/certification_process.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/05_compatibility_certification/certification_process.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/index.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/index.md new file mode 100644 index 0000000000..aa456d0081 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/index.md @@ -0,0 +1,13 @@ +```{include} ../../_shared/05_compatibility_certification/index.md +``` + +```{toctree} +:maxdepth: 1 + +certification_architecture.md +certification_objects.md +certification_levels.md +test_suites.md +certification_process.md +certification_maintenance.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md new file mode 100644 index 0000000000..ded59ab1fa --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md @@ -0,0 +1,6 @@ +--- +tags: [flagos_cloud] +--- + +```{include} ../../_shared/05_compatibility_certification/test_suites_cloud.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/ai_distribution.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/ai_distribution.md new file mode 100644 index 0000000000..6595d32435 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/ai_distribution.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/06_distribution_validation/ai_distribution.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md new file mode 100644 index 0000000000..400496c7dc --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/06_distribution_validation/compatibility_declaration.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/index.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/index.md new file mode 100644 index 0000000000..9be8478895 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/index.md @@ -0,0 +1,12 @@ +```{include} ../../_shared/06_distribution_validation/index.md +``` + +```{toctree} +:maxdepth: 1 + +validation_levels.md +os_distribution.md +ai_distribution.md +joint_validation.md +compatibility_declaration.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md new file mode 100644 index 0000000000..665e7f9a72 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/06_distribution_validation/joint_validation.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/os_distribution.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/os_distribution.md new file mode 100644 index 0000000000..ebd139e2ee --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/os_distribution.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/06_distribution_validation/os_distribution.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/validation_levels.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/validation_levels.md new file mode 100644 index 0000000000..43c322298c --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/06_distribution_validation/validation_levels.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/06_distribution_validation/validation_levels.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/07_quality_metrics/index.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/07_quality_metrics/index.md new file mode 100644 index 0000000000..4dc520a9d9 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/07_quality_metrics/index.md @@ -0,0 +1,9 @@ +--- +tags: [flagos_cloud] +--- + +```{include} ../../_shared/07_quality_metrics/index.md +``` + +```{include} ../../_shared/07_quality_metrics/cloud.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md new file mode 100644 index 0000000000..758f43aec2 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/08_tools_and_skills/skill_catalog.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/09_promotion_criteria/advanced_quality_requirements.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/09_promotion_criteria/advanced_quality_requirements.md new file mode 100644 index 0000000000..9655902461 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/09_promotion_criteria/advanced_quality_requirements.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/09_promotion_criteria/advanced_quality_requirements.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/09_promotion_criteria/index.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/09_promotion_criteria/index.md new file mode 100644 index 0000000000..f2f4ae257c --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/09_promotion_criteria/index.md @@ -0,0 +1,10 @@ +```{include} ../../_shared/09_promotion_criteria/index.md +``` + +```{toctree} +:maxdepth: 1 + +promotion_to_intermediate.md +promotion_to_advanced.md +advanced_quality_requirements.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_advanced.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_advanced.md new file mode 100644 index 0000000000..6386474681 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_advanced.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/09_promotion_criteria/promotion_to_advanced.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_intermediate.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_intermediate.md new file mode 100644 index 0000000000..7f9f843f28 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_intermediate.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/09_promotion_criteria/promotion_to_intermediate.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md b/docs/flagos_homepage/chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md new file mode 100644 index 0000000000..0276fc86ae --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/10_risks_and_maintenance/index.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/cloud_adaptation_guide_index.md b/docs/flagos_homepage/chip_adaptation_guide/cloud_adaptation_guide_index.md new file mode 100644 index 0000000000..8e54b90a52 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/cloud_adaptation_guide_index.md @@ -0,0 +1,21 @@ +# FlagOS Southbound Chip Adaptation Guide: Cloud Chips + +This guide is for cloud AI chip vendors, FlagOS adaptation engineers, framework developers, and distribution partners. It describes the complete path for integrating cloud chips with FlagOS, including cooperation levels, hardware resources, the base environment, compiler and operator adaptation, communication and training/inference framework adaptation, compatibility certification, distribution validation, quality metrics, automation tools, promotion criteria, and risk management. + +Cloud adaptation focuses on multi-card and multi-node operation, cluster communication, heterogeneous interoperability, and large-model training and inference. + +```{toctree} +:maxdepth: 2 +:numbered: 2 + +cloud/01_technical_stack/index.md +cloud/02_cooperation_model/index.md +cloud/03_hardware_resources/index.md +cloud/04_adaptation_plan/index.md +cloud/05_compatibility_certification/index.md +cloud/06_distribution_validation/index.md +cloud/07_quality_metrics/index.md +cloud/08_tools_and_skills/skill_catalog.md +cloud/09_promotion_criteria/index.md +cloud/10_risks_and_maintenance/index.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/01_technical_stack/index.md b/docs/flagos_homepage/chip_adaptation_guide/edge/01_technical_stack/index.md new file mode 100644 index 0000000000..e93c811754 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/01_technical_stack/index.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/01_technical_stack/index.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/02_cooperation_model/index.md b/docs/flagos_homepage/chip_adaptation_guide/edge/02_cooperation_model/index.md new file mode 100644 index 0000000000..f4a8401c5e --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/02_cooperation_model/index.md @@ -0,0 +1,9 @@ +--- +tags: [flagos_edge] +--- + +```{include} ../../_shared/02_cooperation_model/index.md +``` + +```{include} ../../_shared/02_cooperation_model/edge.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/03_hardware_resources/common_constraints.md b/docs/flagos_homepage/chip_adaptation_guide/edge/03_hardware_resources/common_constraints.md new file mode 100644 index 0000000000..c70df1560b --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/03_hardware_resources/common_constraints.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/03_hardware_resources/common_constraints.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/03_hardware_resources/device_evolution.md b/docs/flagos_homepage/chip_adaptation_guide/edge/03_hardware_resources/device_evolution.md new file mode 100644 index 0000000000..aa82bec605 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/03_hardware_resources/device_evolution.md @@ -0,0 +1,6 @@ +--- +tags: [flagos_edge] +--- + +```{include} ../../_shared/03_hardware_resources/device_evolution.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/03_hardware_resources/index.md b/docs/flagos_homepage/chip_adaptation_guide/edge/03_hardware_resources/index.md new file mode 100644 index 0000000000..6975c12357 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/03_hardware_resources/index.md @@ -0,0 +1,16 @@ +--- +tags: [flagos_edge] +--- + +```{include} ../../_shared/03_hardware_resources/index.md +``` + +```{include} ../../_shared/03_hardware_resources/edge.md +``` + +```{toctree} +:maxdepth: 1 + +device_evolution.md +common_constraints.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/index.md b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/index.md new file mode 100644 index 0000000000..cd6f250a32 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/index.md @@ -0,0 +1,19 @@ +--- +tags: [flagos_edge] +--- + +```{include} ../../_shared/04_adaptation_plan/index.md +``` + +```{toctree} +:maxdepth: 1 + +progress_overview.md +stage_00_business.md +stage_01_environment.md +stage_02_flagtree.md +stage_03_flaggems.md +stage_05_validation.md +stage_05_release.md +stage_06_image_build.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/progress_overview.md b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/progress_overview.md new file mode 100644 index 0000000000..b22bc74b69 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/progress_overview.md @@ -0,0 +1,6 @@ +--- +tags: [flagos_edge] +--- + +```{include} ../../_shared/04_adaptation_plan/progress_overview.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_00_business.md b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_00_business.md new file mode 100644 index 0000000000..26e651621d --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_00_business.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/04_adaptation_plan/stage_00_business.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_01_environment.md b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_01_environment.md new file mode 100644 index 0000000000..dffbea5268 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_01_environment.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/04_adaptation_plan/stage_01_environment.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_02_flagtree.md b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_02_flagtree.md new file mode 100644 index 0000000000..d73e54383c --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_02_flagtree.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/04_adaptation_plan/stage_02_flagtree.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_03_flaggems.md b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_03_flaggems.md new file mode 100644 index 0000000000..1a046aecf8 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_03_flaggems.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/04_adaptation_plan/stage_03_flaggems.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_release.md b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_release.md new file mode 100644 index 0000000000..b61a600663 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_release.md @@ -0,0 +1,6 @@ +--- +tags: [flagos_edge] +--- + +```{include} ../../_shared/04_adaptation_plan/stage_05_release_edge.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_validation.md b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_validation.md new file mode 100644 index 0000000000..98aed485c8 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_validation.md @@ -0,0 +1,6 @@ +--- +tags: [flagos_edge] +--- + +```{include} ../../_shared/04_adaptation_plan/stage_05_validation_edge.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.md b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.md new file mode 100644 index 0000000000..f4f3bc4683 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.md @@ -0,0 +1,10 @@ +# Stage 6: Image Build + +Image builds are divided into three levels: base image, runtime image, and application image. + +Follow these documentation guides: + +- [Release Info Overview](https://flagos-ai.github.io/release-info/overview/) +- [Release Info Onboarding Guide](https://flagos-ai.github.io/release-info/contribution/onboarding/) + +If you would like to join, submit an issue in the [release-info repository](https://github.com/flagos-ai/release-info/issues). diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_architecture.md b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_architecture.md new file mode 100644 index 0000000000..5609928de8 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_architecture.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/05_compatibility_certification/certification_architecture.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md new file mode 100644 index 0000000000..03fe141aca --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/05_compatibility_certification/certification_levels.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_maintenance.md b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_maintenance.md new file mode 100644 index 0000000000..d55723d1f7 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_maintenance.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/05_compatibility_certification/certification_maintenance.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md new file mode 100644 index 0000000000..84d2927749 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/05_compatibility_certification/certification_objects.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_process.md b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_process.md new file mode 100644 index 0000000000..f4bad71ad0 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/certification_process.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/05_compatibility_certification/certification_process.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/index.md b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/index.md new file mode 100644 index 0000000000..aa456d0081 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/index.md @@ -0,0 +1,13 @@ +```{include} ../../_shared/05_compatibility_certification/index.md +``` + +```{toctree} +:maxdepth: 1 + +certification_architecture.md +certification_objects.md +certification_levels.md +test_suites.md +certification_process.md +certification_maintenance.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md new file mode 100644 index 0000000000..bc2461b2ab --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md @@ -0,0 +1,6 @@ +--- +tags: [flagos_edge] +--- + +```{include} ../../_shared/05_compatibility_certification/test_suites_edge.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/ai_distribution.md b/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/ai_distribution.md new file mode 100644 index 0000000000..6595d32435 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/ai_distribution.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/06_distribution_validation/ai_distribution.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md b/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md new file mode 100644 index 0000000000..400496c7dc --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/06_distribution_validation/compatibility_declaration.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/index.md b/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/index.md new file mode 100644 index 0000000000..9be8478895 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/index.md @@ -0,0 +1,12 @@ +```{include} ../../_shared/06_distribution_validation/index.md +``` + +```{toctree} +:maxdepth: 1 + +validation_levels.md +os_distribution.md +ai_distribution.md +joint_validation.md +compatibility_declaration.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md b/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md new file mode 100644 index 0000000000..665e7f9a72 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/06_distribution_validation/joint_validation.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/os_distribution.md b/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/os_distribution.md new file mode 100644 index 0000000000..ebd139e2ee --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/os_distribution.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/06_distribution_validation/os_distribution.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/validation_levels.md b/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/validation_levels.md new file mode 100644 index 0000000000..43c322298c --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/06_distribution_validation/validation_levels.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/06_distribution_validation/validation_levels.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/07_quality_metrics/index.md b/docs/flagos_homepage/chip_adaptation_guide/edge/07_quality_metrics/index.md new file mode 100644 index 0000000000..4a6ab82ee0 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/07_quality_metrics/index.md @@ -0,0 +1,9 @@ +--- +tags: [flagos_edge] +--- + +```{include} ../../_shared/07_quality_metrics/index.md +``` + +```{include} ../../_shared/07_quality_metrics/edge.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md b/docs/flagos_homepage/chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md new file mode 100644 index 0000000000..758f43aec2 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/08_tools_and_skills/skill_catalog.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/09_promotion_criteria/advanced_quality_requirements.md b/docs/flagos_homepage/chip_adaptation_guide/edge/09_promotion_criteria/advanced_quality_requirements.md new file mode 100644 index 0000000000..9655902461 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/09_promotion_criteria/advanced_quality_requirements.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/09_promotion_criteria/advanced_quality_requirements.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/09_promotion_criteria/index.md b/docs/flagos_homepage/chip_adaptation_guide/edge/09_promotion_criteria/index.md new file mode 100644 index 0000000000..f2f4ae257c --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/09_promotion_criteria/index.md @@ -0,0 +1,10 @@ +```{include} ../../_shared/09_promotion_criteria/index.md +``` + +```{toctree} +:maxdepth: 1 + +promotion_to_intermediate.md +promotion_to_advanced.md +advanced_quality_requirements.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_advanced.md b/docs/flagos_homepage/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_advanced.md new file mode 100644 index 0000000000..6386474681 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_advanced.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/09_promotion_criteria/promotion_to_advanced.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_intermediate.md b/docs/flagos_homepage/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_intermediate.md new file mode 100644 index 0000000000..7f9f843f28 --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_intermediate.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/09_promotion_criteria/promotion_to_intermediate.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge/10_risks_and_maintenance/index.md b/docs/flagos_homepage/chip_adaptation_guide/edge/10_risks_and_maintenance/index.md new file mode 100644 index 0000000000..0276fc86ae --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge/10_risks_and_maintenance/index.md @@ -0,0 +1,2 @@ +```{include} ../../_shared/10_risks_and_maintenance/index.md +``` diff --git a/docs/flagos_homepage/chip_adaptation_guide/edge_adaptation_guide_index.md b/docs/flagos_homepage/chip_adaptation_guide/edge_adaptation_guide_index.md new file mode 100644 index 0000000000..9a7cb938ec --- /dev/null +++ b/docs/flagos_homepage/chip_adaptation_guide/edge_adaptation_guide_index.md @@ -0,0 +1,21 @@ +# FlagOS Southbound Chip Adaptation Guide: Edge Chips + +This guide is for edge AI chip vendors, FlagOS adaptation engineers, system software developers, and distribution partners. It describes the complete path for integrating edge chips with FlagOS, including cooperation levels, device investment, the base environment, the FlagTree compiler, FlagGems operators, compatibility certification, distribution validation, quality metrics, automation tools, promotion criteria, and risk management. + +Edge adaptation focuses on stable operation on single-node or small-scale devices, resource-constrained environments, driver and firmware coordination, and end-to-end efficiency. Edge adaptation does not require the FlagCX communication library or the FlagScale large-model training and inference framework by default; if a product explicitly supports these capabilities, they can be evaluated separately within the actual cooperation scope. + +```{toctree} +:maxdepth: 2 +:numbered: 2 + +edge/01_technical_stack/index.md +edge/02_cooperation_model/index.md +edge/03_hardware_resources/index.md +edge/04_adaptation_plan/index.md +edge/05_compatibility_certification/index.md +edge/06_distribution_validation/index.md +edge/07_quality_metrics/index.md +edge/08_tools_and_skills/skill_catalog.md +edge/09_promotion_criteria/index.md +edge/10_risks_and_maintenance/index.md +``` diff --git a/docs/flagos_homepage/index-backup.md b/docs/flagos_homepage/index-backup.md deleted file mode 100644 index f9844f0e6e..0000000000 --- a/docs/flagos_homepage/index-backup.md +++ /dev/null @@ -1,182 +0,0 @@ ---- -sd_hide_title: true ---- - -# Documentation - -
-

FlagOS

-

A unified, open-source system software stack designed for a variety of AI chips

- - FlagOS Overview - - - - -
- -
- -
-

FlagGems

-
- A high-performance general-purpose operator library implemented with the Triton programming language and its extended languages. -
- -
- - -
-

FlagTree

-
- An open-source, unified compiler for multiple AI chips. -
- -
- - -
-

FlagScale

-
- A comprehensive toolkit designed to support the entire lifecycle of large models. -
- -
- - -
-

FlagCX

-
- A scalable and adaptive unified communication library for cross-chip environments. -
- -
- - -
-

Megatron-LM-FL

-
- A fork of Megatron-LM that introduces a plugin-based architecture for supporting diverse AI chips, built on top of FlagOS, a unified open-source AI system software stack. -
- -
- - -
-

vllm-plugin-FL

-
- A plugin for the vLLM inference/serving framework, built on FlagOS's unified multi-chip backend — including the unified operator library FlagGems and the unified communication library FlagCX. -
- -
- - -
-

TransformerEngine-FL

-
- A fork of TransformerEngine that introduces a plugin-based architecture for supporting diverse AI chips, built on top of FlagOS, a unified open-source AI system software stack.. -
- -
- - -
-

verl-FL

-
- A fork of [veRL](https://github.com/verl-project/verl) (Volcano Engine Reinforcement Learning for LLMs) that extends the upstream library with multi-chip/multi-hardware support via the FlagOS ecosystem. -
- -
- - -
-

FlagOS-Robo

-
- An integrated training and inference framework for AI models used in robots, so-called Embodied Intelligence. It is built upon the unified and open-source AI system software stack, FlagOS, which supports various AI chips. -
- -
- - -
-

KernelGen

-
- An operator auto-generation tool. -
- -
- - -
-

FlagRelease

-
- A platform dedicated to the automatic migration, adaptation and release of large models for multi-architecture AI chips. -
- -
- - -
-

FlagPerf

-
- An integrated AI hardware evaluation engine. -
- -
- - -
-

Online Laboratory

-
- An online laboratory providing cloud-based development environments. -
- -
- - -
-

FlagOS Skills

-
- Compatible with **Claude Code**, **Cursor**, **Codex**, and any agent supporting the [Agent Skills standard](https://agentskills.io/specification).. -
- -
- -
-

Start to Use FlagOS

-

Join us to co-build an open AI chip development ecosystem

- - - - - FlagOS Homepage - -
diff --git a/docs/flagos_homepage/index.md b/docs/flagos_homepage/index.md index 9da368cb1c..1b61bf5eeb 100644 --- a/docs/flagos_homepage/index.md +++ b/docs/flagos_homepage/index.md @@ -12,8 +12,18 @@ sd_hide_title: true A unified, open-source system software stack designed for a variety of AI chips [FlagOS Overview](overview.md){ .flagos-outline-btn } +[Cloud Chip Adaptation Guide](chip_adaptation_guide/cloud_adaptation_guide_index.md){ .flagos-outline-btn } +[Edge Chip Adaptation Guide](chip_adaptation_guide/edge_adaptation_guide_index.md){ .flagos-outline-btn } ::: +```{toctree} +:maxdepth: 1 +:class: flagos-guide-root-toctree + +chip_adaptation_guide/cloud_adaptation_guide_index.md +chip_adaptation_guide/edge_adaptation_guide_index.md +``` + ## FlagOS Core Libraries ````{grid} 1 1 1 1 @@ -289,4 +299,4 @@ A CI/CD toolchain that streamlines large-model development across diverse AI chi Join us to co-build an open AI chip development ecosystem [FlagOS Homepage](https://flagos.io/){ .btn .btn-primary .btn-lg } -::: \ No newline at end of file +::: diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/01_technical_stack/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/01_technical_stack/index.mo new file mode 100644 index 0000000000..11d0eb3ccd Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/01_technical_stack/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/01_technical_stack/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/01_technical_stack/index.po new file mode 100644 index 0000000000..5aaf5f8a85 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/01_technical_stack/index.po @@ -0,0 +1,36 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/01_technical_stack/index.md:2 +msgid "FlagOS Technology Stack Overview" +msgstr "FlagOS 技术堆栈全景" + +#: ../../chip_adaptation_guide/_shared/01_technical_stack/index.md:4 +msgid "The figure below shows the latest FlagOS technology stack. The FlagOS ecosystem currently includes three categories of open-source projects. The FlagOS open-source core libraries include the FlagGems general-purpose large-model and domain-specific operator libraries, the FlagScale training and inference framework, the FlagTree unified compiler, and the FlagCX unified communication library. We also define ecosystem-enabling projects, including plugins for training and inference engines and open-source tools designed by the FlagOS community, such as KernelGen for automatic operator generation, FlagRelease for automatic migration and release, and FlagPerf for multi-chip evaluation." +msgstr "下图展示 FlagOS 最新技术堆栈全景。FlagOS生态目前包括三大开源项目门类,分别为FlagOS开源核心库:包括了FlagGems通用大模型算子库,多领域算子库;FlagScale训练推理框架,FlagTree统一编译器和FlagCX统一通信库。同时我们定义了FlagOS生态使能项目,对接训练推理引擎的插件,以及FlagOS社区主导设计的开源工具,包括KernelGen算子自动生成,FlagRelease自动迁移发版和FlagPerf多芯评测工具。" + +#: ../../chip_adaptation_guide/_shared/01_technical_stack/index.md:6 +msgid "![image.png](../../../images/flagos-architecture-en.png)" +msgstr "![image.png](../../../images/flagos-architecture-zh.png)" + +#: ../../chip_adaptation_guide/_shared/01_technical_stack/index.md:6 +msgid "image.png" +msgstr "image.png" + +#: ../../chip_adaptation_guide/_shared/01_technical_stack/index.md:8 +msgid "Figure 1. FlagOS technology stack overview" +msgstr "图 1 FlagOS 技术堆栈全景" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/02_cooperation_model/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/02_cooperation_model/index.mo new file mode 100644 index 0000000000..bc09bfd9f6 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/02_cooperation_model/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/02_cooperation_model/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/02_cooperation_model/index.po new file mode 100644 index 0000000000..4475db6582 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/02_cooperation_model/index.po @@ -0,0 +1,179 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-20 00:00+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/02_cooperation_model/cloud.md:2 +msgid "Cooperation Level Model" +msgstr "合作分级模型" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Cooperation is divided into three levels: basic, intermediate, and advanced. Chip companies progress through the levels step by step." +msgstr "合作分为初级、中级、高级三个级别,芯片公司逐级晋升。" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "**Dimension**" +msgstr "**维度**" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "**Basic (integration validation)**" +msgstr "**初级(接入验证)**" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "**Intermediate (deep adaptation)**" +msgstr "**中级(深度适配)**" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "**Advanced (ecosystem co-building)**" +msgstr "**高级(生态共建)**" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Positioning" +msgstr "定位" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Integration validation" +msgstr "接入验证" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Deep adaptation" +msgstr "深度适配" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Ecosystem co-building" +msgstr "生态共建" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Operators" +msgstr "算子" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Adopt selected high-performance FlagGems operators" +msgstr "采纳部分 FlagGems 高性能算子" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Fully support the FlagGems Stable operator set and commit to using FlagGems by default" +msgstr "芯片全面支持 FlagGems 稳定(Stable)算子集合,承诺默认采用 FlagGems" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Co-build FlagGems and generate specialized operators with KernelGen" +msgstr "参与 FlagGems 共建,KernelGen 算子特化生成" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Compiler" +msgstr "编译器" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Use the FlagTree compiler and C++ Wrapper" +msgstr "采用 FlagTree 编译器 + C++ Wrapper" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Optimize with Hints and TLE; use FlagTree as the default compiler" +msgstr "采用 Hints 机制优化 + TLE,默认使用FlagTree 作为编译器" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Co-build all three Triton TLE implementation layers" +msgstr "全面共建Triton TLE三层实现路径" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Training/inference framework" +msgstr "训推框架" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Adapt the FlagScale plugin" +msgstr "适配 FlagScale 插件" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Model training and inference" +msgstr "模型训练 + 推理" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Participate in large-model T+0 releases" +msgstr "参与大模型 T+0 发布" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Communication" +msgstr "通信" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "—" +msgstr "—" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Adapt FlagCX" +msgstr "适配 FlagCX" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Enable FlagCX by default and continuously maintain the adaptation code" +msgstr "默认启用 FlagCX,持续维护适配代码" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "CI/CD" +msgstr "CI/CD" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Use the FlagOS CI resource pool for unit tests" +msgstr "支持使用 FlagOS CI 资源池执行单元测试" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Co-build test cases, upstream code, and provide machine resources" +msgstr "共建测试用例,代码进主线,提供机器资源" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Bring in operating-system and cloud partner communities
" +msgstr "带入操作系统和云合作社区
" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Compatibility and security" +msgstr "兼容性与安全" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Provide base container image materials and download channels for customized third-party dependencies" +msgstr "提供基础容器镜像物料;提供定制第三方依赖包下载渠道" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Obtain FlagOS certification, promptly update customized packages, and respond immediately to package adaptation issues" +msgstr "设备获得 FlagOS 认证,及时升级定制软件包;对软件包适配问题承诺第一时间响应" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Share security working-group results and bring them into downstream distributions" +msgstr "共享安全工作组成果,下游发行版带入" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Open-source community" +msgstr "开源社区" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Basic member" +msgstr "初级会员" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Intermediate member, eligible to run for PMC" +msgstr "中级会员,可竞选 PMC" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Advanced member, with course and case-study coverage" +msgstr "高级会员,课程/案例覆盖" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Business expansion" +msgstr "商业拓展" + +#: ../../chip_adaptation_guide/cloud/02_cooperation_model/index.md:8 +msgid "Bring in partner clouds and MaaS platforms" +msgstr "带入合作云、MaaS 平台" + +msgid "Bring in upstream and downstream ecosystem communities" +msgstr "上下游社区生态带入" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/common_constraints.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/common_constraints.mo new file mode 100644 index 0000000000..a4be5174ca Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/common_constraints.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/common_constraints.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/common_constraints.po new file mode 100644 index 0000000000..03973c6058 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/common_constraints.po @@ -0,0 +1,36 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/common_constraints.md:2 +msgid "Key Constraints" +msgstr "关键约束说明" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/common_constraints.md:4 +msgid "CI devices must stay online 24/7 and be managed through the unified FlagOS CI/CD resource pool." +msgstr "CI 设备要求 7×24 小时在线,纳入 FlagOS CI/CD 资源池统一调度。" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/common_constraints.md:6 +msgid "SVT devices should be physically isolated from CI devices to prevent validation jobs and CI jobs from competing for resources." +msgstr "SVT 设备应与 CI 设备物理隔离,避免验证任务与 CI 任务争抢资源。" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/common_constraints.md:8 +msgid "The dedicated FlagRelease machine should be configured independently and should not be shared with other workloads." +msgstr "FlagRelease 专机建议独立配置,不与其他工作负载混用。" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/common_constraints.md:10 +msgid "All devices must support automated access. Driver and SDK packages should preferably be available through open downloads or long-lived authorization." +msgstr "所有设备需满足自动化获取要求,驱动/SDK 获取方式优先开放下载或长期有效授权。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/device_evolution.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/device_evolution.mo new file mode 100644 index 0000000000..e9b6907ca0 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/device_evolution.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/device_evolution.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/device_evolution.po new file mode 100644 index 0000000000..f753e83cef --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/device_evolution.po @@ -0,0 +1,36 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md:2 +msgid "Device Investment Evolution" +msgstr "设备投入演进图" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md:5 +msgid "![image.png](/chip_adaptation_guide/assets/cloud/device_evolution-en.png)" +msgstr "![image.png](/chip_adaptation_guide/assets/cloud/device_evolution.png)" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md:5 ../../chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md:9 +msgid "image.png" +msgstr "image.png" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md:9 +msgid "![image.png](/chip_adaptation_guide/assets/edge/device_evolution-en.png)" +msgstr "![image.png](/chip_adaptation_guide/assets/edge/device_evolution.png)" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md:12 +msgid "Figure 2. Device investment evolution" +msgstr "图 2 设备投入演进图" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/index.mo new file mode 100644 index 0000000000..5e8129e746 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/index.po new file mode 100644 index 0000000000..67f4a84a0b --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/03_hardware_resources/index.po @@ -0,0 +1,240 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/index.md:2 +msgid "Hardware Resource Investment Plan" +msgstr "硬件资源投入计划" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/cloud.md:2 +msgid "Throughout chip adaptation, the chip company must provide physical devices at each stage to support CI/CD, validation, and release. For the first two batches, two machines are recommended for each batch. KernelGen is expected to use two machines so that operators can be specialized, tested in batches, and used for operator competitions. Single-card, four-card, or eight-card configurations should be confirmed according to the scenario. Additional machines are usually required for Day 0, distribution, cloud-native, or showroom plans. Virtualization can supplement multi-OS validation, but critical validation still requires physical devices." +msgstr "芯片适配全过程中,芯片公司需按阶段投入物理设备以支撑 CI/CD、验证和发布。资源补充说明:第一批建议 2 台机器,第二批建议 2 台机器;KernelGen 建议 2 台机器,可以批量特化和测试算子,以及支持算子比赛;单卡、4 卡或 8 卡按具体场景确认。进入 Day 0、发行版、云原生或展厅展示计划时,通常需要额外机器。针对多 OS 操作系统验证,虚拟化技术可以作为补充,但关键验证仍需物理设备。" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "**Stage ID**" +msgstr "**阶段标识**" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "**Stage**" +msgstr "**阶段**" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "**Required devices**" +msgstr "**所需设备**" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "**Purpose**" +msgstr "**用途**" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "**Configuration requirements**" +msgstr "**配置要求**" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "**Availability timing**" +msgstr "**投入时间点**" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Stage 1" +msgstr "阶段 1" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Environment readiness" +msgstr "环境就绪" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "1 chip development board or server with remote access for the FlagOS team" +msgstr "芯片开发板/服务器,供 FlagOS 远程访问" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Single-card operation" +msgstr "单卡可运行,Ubuntu OS,64GB+ 内存" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Ubuntu OS, 64 GB or more memory" +msgstr "单卡可运行,Ubuntu OS,64GB+ 内存" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Prepare immediately after the Stage 0 MOU is signed" +msgstr "阶段 0 签署 MOU 后立即准备" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Stage 2" +msgstr "阶段 2" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "FlagTree compiler integration" +msgstr "FlagTree 编译器接入" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "1 device added to the CI resource pool" +msgstr "1 台 → 纳入 CI 资源池" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Compiler build validation and CI pipeline triggers" +msgstr "编译器构建验证、CI 流水线触发" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Single card, 200 GB or more disk space, network access to the FlagOS CI cluster" +msgstr "单卡,200GB+ 磁盘,网络可达 FlagOS CI 集群" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Connect to CI when Stage 2 starts" +msgstr "阶段 2 启动时接入 CI" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Stage 3" +msgstr "阶段 3" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "FlagGems operator adaptation" +msgstr "FlagGems 算子适配" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "1 CI machine and 1 debug machine" +msgstr "CI 1 台 + 调试机 1 台" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Automated CI tests and operator development/debugging" +msgstr "CI 自动化测试 + 算子开发调试" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "The CI machine must stay reliably online; the vendor may keep the debug machine locally" +msgstr "CI 机稳定在线,调试机可由厂商保留本地" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Confirm the CI machine SLA when Stage 3 starts" +msgstr "阶段 3 启动时确认 CI 机 SLA" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Stage 4a" +msgstr "阶段 4a" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "FlagCX communication adaptation" +msgstr "FlagCX 通信适配" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "At least 2 machines, each with at least 4 cards" +msgstr "至少 2 台(每台 ≥4 卡)" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Collective communication tests and multi-node communication validation" +msgstr "集合通信测试、多机通信验证" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "High-speed interconnect such as IB or RoCE; homogeneous cluster" +msgstr "高速互联(IB/RoCE),同构集群" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Available when Stage 4a starts" +msgstr "阶段 4a 启动时到位" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Stage 4b" +msgstr "阶段 4b" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "FlagScale training and inference" +msgstr "FlagScale 训练推理" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "At least 2 machines, reused from Stage 4a" +msgstr "至少 2 台(复用 4a)" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Distributed training and inference service validation" +msgstr "分布式训练、推理服务验证" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Same as Stage 4a, with persistent storage mounted" +msgstr "同上,需持久化存储挂载" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Reuse Stage 4a devices" +msgstr "复用阶段 4a 设备" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Stage 5" +msgstr "阶段 5" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "End-to-end validation" +msgstr "全流程验证" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "1 CI machine and 2 SVT machines" +msgstr "CI 1 台 + SVT 2 台" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Continuous CI regression and system validation testing" +msgstr "CI 持续回归 + 系统验证测试" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "SVT machines must be isolated from CI machines to avoid interference" +msgstr "SVT 机需与 CI 机隔离,避免相互干扰" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Available when Stage 5 starts" +msgstr "阶段 6 启动时到位" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Stage 6" +msgstr "阶段 6" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "FlagRelease release" +msgstr "FlagRelease 发布" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "1 dedicated FR machine" +msgstr "1 台 FR 专机" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Model migration, image builds, and release validation" +msgstr "模型迁移、镜像构建、发布验证" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Single card or above, 200 GB or more disk space, high-speed network" +msgstr "单卡及以上,200GB+ 磁盘,高速网络" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Available when Stage 6 starts" +msgstr "阶段 6 启动时到位" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Stage 7" +msgstr "阶段 7" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "KernelGen" +msgstr "KernelGen" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "2 machines" +msgstr "2 台" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Automatic operator generation, tuning, and operator competitions" +msgstr "算子自动生成调优,算子比赛" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "Single-card operation with FlagTree compiler and operator development/tuning environments installed" +msgstr "单卡可运行,安装 FlagTree 编译器等环境,需要提供算子开发和调优的代码和文档等" + +#: ../../chip_adaptation_guide/cloud/03_hardware_resources/index.md:8 +msgid "After Stage 2 is complete" +msgstr "阶段 2 完成" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/index.mo new file mode 100644 index 0000000000..892dfb5305 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/index.po new file mode 100644 index 0000000000..b7d94993bc --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/index.po @@ -0,0 +1,27 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-20 00:00+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/index.md:5 +msgid "Adaptation Stages and Execution Plan" +msgstr "适配阶段与执行计划" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/index.md:9 +msgid "The cloud chip path progresses through six stages from kickoff to advanced adaptation." +msgstr "芯片从起步到完成高级适配,按六个阶段推进。" + +msgid "The edge chip path progresses through the stages from kickoff to advanced adaptation." +msgstr "芯片从起步到完成高级适配,按阶段推进。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/progress_overview.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/progress_overview.mo new file mode 100644 index 0000000000..f2f481faa0 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/progress_overview.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/progress_overview.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/progress_overview.po new file mode 100644 index 0000000000..e4d97812dc --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/progress_overview.po @@ -0,0 +1,64 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:2 +msgid "Overall Progress Overview" +msgstr "全阶段进度总览" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:5 +msgid "**Overall progress overview:** The figure below shows the complete path from contract kickoff, environment readiness, compiler integration, operator coverage, communication, training and inference, validation, compliance, release, and promotion. It is organized around key deliverables, cooperation-level evolution, and hardware investment." +msgstr "**全阶段进度总览说明:** 下图从关键成果、合作等级演进和硬件投入三个维度,展示芯片从签约启动、环境就绪、编译器接入、算子覆盖、通信与训练推理,到验证合规和发布推广的完整路径。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:7 +msgid "**Performance metric requirements:** Starting from Stage 3, operator performance, communication throughput, model accuracy, and end-to-end efficiency gradually become acceptance metrics. For advanced cooperation, performance metrics are a key baseline. Under equivalent compute, the chip must reach a defined percentage of the NVIDIA baseline; the specific percentage will be confirmed later through the KT2 milestone." +msgstr "**性能指标要求:** 从阶段 3 起,算子性能、通信吞吐、模型精度和端到端效率逐步进入指标范围;高级合作会以性能指标作为重要基准,需要再同等算力下达到 NVIDIA 基线 X%,具体比例后续按 KT2期口径确定。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:11 +msgid "**Overall progress overview:** The figure below shows the complete path from contract kickoff, environment readiness, compiler integration, operator coverage, validation, compliance, release, and promotion. It is organized around key deliverables, cooperation-level evolution, and hardware investment." +msgstr "**全阶段进度总览说明:** 下图从关键成果、合作等级演进和硬件投入三个维度,展示芯片从签约启动、环境就绪、编译器接入、算子覆盖,到验证合规和发布推广的完整路径。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:13 +msgid "**Performance metric requirements:** Starting from Stage 3, operator performance, model accuracy, and end-to-end efficiency gradually become acceptance metrics. For advanced cooperation, performance metrics are a key baseline." +msgstr "**性能指标要求:** 从阶段 3 起,算子性能、模型精度和端到端效率逐步进入指标范围;高级合作会以性能指标作为重要基准。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:16 +msgid "**Quality and continuous maintenance requirements:** Vendors must commit to maintaining the mainline adaptation code. FlagOS upgrades and vendor Driver/SDK upgrades must both be covered by CI/CD validation so that performance results remain sustainable, reproducible, and releasable." +msgstr "**质量与持续维护要求:** 厂商需承诺维护主线适配代码,FlagOS 升级与厂商 Driver/SDK 升级需共同纳入 CI/CD 验证,确保性能结果可持续、可复测、可发布。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:19 +msgid "![image.png](/chip_adaptation_guide/assets/cloud/progress_overview-en.png)" +msgstr "![image.png](/chip_adaptation_guide/assets/cloud/progress_overview.png)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:19 ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:23 +msgid "image.png" +msgstr "image.png" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:23 +msgid "![image.png](/chip_adaptation_guide/assets/edge/progress_overview-en.png)" +msgstr "![image.png](/chip_adaptation_guide/assets/edge/progress_overview.png)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:26 +msgid "Figure 3. Overall progress overview" +msgstr "图 3 全阶段进度总览" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:29 +msgid "Estimated total duration: about 5-7 months from signing to release. Total hardware investment increases with the cooperation target. The basic path requires about 1 development board, 1 CI machine, 2 SVT machines, and 1 dedicated FR machine. Additional machines should be added for Day 0, distribution, cloud-native, or showroom plans." +msgstr "预计总周期:约 5-7 个月(从签约到发布)。硬件总投入随合作目标递进:基础路径约 1 台开发板 + 1 台 CI + 2 台 SVT + 1 台 FR 专机;进入 Day 0、发行版、云原生或展厅展示计划时按场景追加。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:33 +msgid "Estimated total duration: about 3-5 months from signing to release. Total hardware investment increases with the cooperation target. The basic path requires about 1 development board, 1 CI machine, 2 SVT machines, and 1 dedicated FlagRelease (FR) machine. Additional machines should be added for Day 0, distribution, cloud-native, or showroom plans." +msgstr "预计总周期:约 3-5个月(从签约到发布)。硬件总投入随合作目标递进:基础路径约 1 台开发板 + 1 台 CI + 2 台 SVT + 1 台 FlagRelease(FR) 专机;进入 Day 0、发行版、云原生或展厅展示计划时按场景追加。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_00_business.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_00_business.mo new file mode 100644 index 0000000000..af53498222 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_00_business.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_00_business.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_00_business.po new file mode 100644 index 0000000000..66a1561009 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_00_business.po @@ -0,0 +1,105 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 15:50+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:2 +msgid "Stage 0: Business Kickoff and Contracting" +msgstr "阶段 0:商务启动与签约" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:4 +msgid "**Goal**" +msgstr "**目标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:6 +msgid "Establish the cooperation relationship and clarify the responsibilities of both parties." +msgstr "确立合作关系,明确双方权责。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:8 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:10 +msgid "The chip company has a base SDK/driver and can provide a development environment." +msgstr "芯片公司具备基础 SDK/驱动,可提供开发环境" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:12 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:14 +msgid "Sign an MOU defining the cooperation goals, adaptation scope, and division of responsibilities." +msgstr "签署 MOU,明确合作目标、适配范围、责任分工" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:15 +msgid "Connect with the FlagOS cooperation manager and assign contacts for both parties." +msgstr "对接 FlagOS 合作经理,分配双方接口人" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:16 +msgid "Set the initial target level for chip adaptation, usually starting at the intermediate level." +msgstr "确定芯片适配的初始目标级别(通常从中级起步)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:17 +msgid "Sign the Contributor License Agreement (CLA)." +msgstr "签署贡献者许可协议(CLA)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:19 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:21 +msgid "Signed MOU and contact list for both parties." +msgstr "已签署的 MOU 文档,双方接口人通讯录" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:23 +msgid "**FlagOS contacts**" +msgstr "**FlagOS 方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:25 +msgid "Dedicated cooperation manager and technical architect." +msgstr "专属合作经理、技术架构师" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:27 +msgid "**Chip company contacts**" +msgstr "**芯片方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:29 +msgid "Project director and technical lead." +msgstr "项目总监、技术负责人" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:31 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:33 +msgid "MOU signed and contacts confirmed by both parties." +msgstr "MOU 签署完成,双方接口人确认" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:35 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:37 +msgid "1-2 weeks" +msgstr "1-2 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:39 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:41 +msgid "MOU template, CLA, and cooperation handbook" +msgstr "MOU 模板、CLA 协议、合作手册" + diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_01_environment.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_01_environment.mo new file mode 100644 index 0000000000..aa6f06fd16 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_01_environment.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_01_environment.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_01_environment.po new file mode 100644 index 0000000000..efe6904827 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_01_environment.po @@ -0,0 +1,142 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:2 +msgid "" +"Stage 1: Development Board Readiness and Base Environment Adaptation " +"(Chip-Level)" +msgstr "阶段 1:开发板就绪与基础环境适配(芯片级)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:4 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:6 +msgid "The MOU has been signed." +msgstr "MOU 已签署" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:8 +msgid "**Hardware investment**" +msgstr "**硬件投入**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:10 +msgid "" +"1 development board or server, with remote access provided by the vendor " +"to the FlagOS team." +msgstr "1 台开发板/服务器,厂商提供远程访问权限给 FlagOS 团队" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:12 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:14 +msgid "" +"Provide the chip development board and select at least one operating " +"system, preferably Ubuntu." +msgstr "提供芯片开发板,选择不少于一个操作系统(以 Ubuntu 为首选)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:15 +msgid "" +"Provide the driver and SDK access method. Open downloads are preferred. " +"If authorization-only downloads are used, programmatic authentication " +"must support CI/CD automation." +msgstr "提供驱动及 SDK 获取方式(推荐开放下载;若仅支持授权下载,需支持 CI/CD 自动化场景下的程序化认证)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:16 +msgid "" +"Confirm base image compatibility. A neutral image is preferred, followed " +"by a vendor-provided image." +msgstr "确认基础镜像兼容性(优先中立镜像,其次厂商提供)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:17 +msgid "" +"Confirm software compatibility, including kernel, Python, and PyTorch " +"community version coverage." +msgstr "确认软件兼容性:内核、Python、PyTorch 社区版本范围覆盖" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:18 +msgid "" +"Validate the base environment: PyTorch can recognize the target device " +"backend, and the equivalent device discovery API returns an available " +"device." +msgstr "基础环境验证:PyTorch 可识别目标设备后端,等效设备发现接口返回可用" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:20 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:22 +msgid "Environment readiness report and driver/SDK access documentation." +msgstr "环境就绪确认报告,驱动/SDK 获取方式文档" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:24 +msgid "**FlagOS contacts**" +msgstr "**FlagOS 方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:26 +msgid "Platform engineering team." +msgstr "平台工程团队" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:28 +msgid "**Chip company contacts**" +msgstr "**芯片方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:30 +msgid "Driver and firmware engineers." +msgstr "驱动/固件工程师" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:32 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:34 +msgid "" +"The chip can be correctly identified by the OS, for example through lspci" +" or dmidecode." +msgstr "芯片可被 OS 正确识别(lspci / dmidecode 等)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:35 +msgid "PyTorch can recognize the chip backend." +msgstr "PyTorch 可识别芯片后端" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:36 +msgid "" +"Basic operators, such as matrix multiplication and elementwise " +"operations, execute correctly." +msgstr "基础算子(矩阵乘、元素级操作)可正常执行" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:37 +msgid "The driver can be obtained unattended." +msgstr "驱动可无人值守获取" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:39 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:41 +msgid "1-2 weeks" +msgstr "1-2 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:43 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:45 +msgid "" +"Environment configuration guide, driver installation manual, and " +"compatibility matrix." +msgstr "环境配置指南、驱动安装手册、兼容性矩阵" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_02_flagtree.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_02_flagtree.mo new file mode 100644 index 0000000000..0cff19e226 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_02_flagtree.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_02_flagtree.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_02_flagtree.po new file mode 100644 index 0000000000..55eef106de --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_02_flagtree.po @@ -0,0 +1,157 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:2 +msgid "" +"Stage 2: Compiler Adaptation - FlagTree Branch Integration (Entry-Level " +"Threshold)" +msgstr "阶段 2:编译器适配 — FlagTree 分支接入(初级门槛)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:4 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:6 +msgid "" +"The development board environment is ready, and the chip Triton backend " +"or compatibility layer is available." +msgstr "开发板环境就绪,芯片 Triton 后端(或兼容层)可用" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:8 +msgid "**Hardware investment**" +msgstr "**硬件投入**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:10 +msgid "" +"Add the Stage 1 device to the FlagOS CI resource pool and keep it online " +"24/7." +msgstr "阶段 1 的 1 台设备纳入 FlagOS CI 资源池,要求 7×24 在线" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:12 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:14 +msgid "Create a protected chip branch in the FlagTree repository." +msgstr "在 FlagTree 仓库中创建芯片受保护分支" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:15 +msgid "" +"The chip company provides adaptation code or a compatibility layer for " +"the Triton backend." +msgstr "芯片公司提供 Triton 后端的适配代码或兼容层" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:16 +msgid "" +"Integrate a C++ Wrapper if a bridge from Triton IR to the vendor SDK is " +"required." +msgstr "集成 C++ Wrapper(若需要桥接 Triton IR 到厂商 SDK)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:17 +msgid "Complete instruction mapping through the unified FLIR IR layer." +msgstr "通过 FLIR 统一 IR 层完成指令映射" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:18 +msgid "Configure a basic CI pipeline for automatic builds and compilation tests." +msgstr "配置基础 CI 流水线:自动构建 + 编译测试" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:20 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:22 +msgid "Available FlagTree chip branch and running basic CI pipeline." +msgstr "FlagTree 芯片分支可用,基础 CI 流水线运行" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:24 +msgid "**FlagOS contacts**" +msgstr "**FlagOS 方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:26 +msgid "FlagTree compiler team." +msgstr "FlagTree 编译器团队" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:28 +msgid "**Chip company contacts**" +msgstr "**芯片方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:30 +msgid "Compiler and toolchain engineers." +msgstr "编译器/工具链工程师" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:32 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:34 +msgid "Triton kernels can be compiled successfully to the chip backend." +msgstr "Triton kernel 可成功编译到芯片后端" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:35 +msgid "At least 10 basic operators can be compiled and executed correctly." +msgstr "至少 10 个基础算子可正确编译和执行" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:36 +msgid "" +"CI build pass rate is at least 95%, and machine availability is at least " +"99%." +msgstr "CI 构建通过率 ≥95%,机器可用性 ≥99%" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:38 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:40 +msgid "2-4 weeks" +msgstr "2-4 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:42 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:44 +msgid "" +"[FlagTree integration " +"guide](https://jwolpxeehx.feishu.cn/docx/VIridPa16odD9hxViDLcsD2vnXb): " +"integration plans for GPGPU and DSA/NPU devices.
Triton-TLE: [GitHub " +"documentation](https://github.com/flagos-ai/FlagTree/wiki/TLE), " +"[reference NV implementation](https://github.com/flagos-" +"ai/FlagTree/tree/triton_v3.6.x), [TLE vendor PR " +"template](https://jwolpxeehx.feishu.cn/file/VIOsbkSYZoI3I5xhxdLcuz8snVg?from=from_copylink)," +" [TLE raw CUDA integration vendor " +"example](https://jwolpxeehx.feishu.cn/wiki/Ipvbwa1WYintOwkmNzUchJJun0c).
C++" +" Wrapper: [libtriton_jit documentation](https://github.com/flagos-" +"ai/libtriton_jit), [sample PR](https://github.com/flagos-" +"ai/libtriton_jit/pull/12/changes), [C++ Runtime multi-backend and Triton " +"operator development " +"sharing](https://jwolpxeehx.feishu.cn/slides/RPkms0jYol2TA7dvbfScjVJWnlu?from=from_copylink)." +msgstr "" +"[FlagTree " +"接入指南](https://jwolpxeehx.feishu.cn/docx/VIridPa16odD9hxViDLcsD2vnXb): 包含 " +"GPGPU、DSA/NPU 两类设备的接入方案
Triton-TLE:[GitHub(说明文档)](https://github.com" +"/flagos-ai/FlagTree/wiki/TLE) [参考 NV 具体实现的代码](https://github.com/flagos-" +"ai/FlagTree/tree/triton_v3.6.x) [TLE 厂商 PR " +"模版](https://jwolpxeehx.feishu.cn/file/VIOsbkSYZoI3I5xhxdLcuz8snVg?from=from_copylink)" +" [TLE-Raw CUDA " +"接入-厂商示例说明](https://jwolpxeehx.feishu.cn/wiki/Ipvbwa1WYintOwkmNzUchJJun0c)
C++" +" Wrapper:[libtriton_jit 说明文档](https://github.com/flagos-ai/libtriton_jit)" +" [寒武纪示例](https://github.com/flagos-ai/libtriton_jit/pull/12/changes) [C++" +" Runtime 多后端与 Triton " +"算子开发分享](https://jwolpxeehx.feishu.cn/slides/RPkms0jYol2TA7dvbfScjVJWnlu?from=from_copylink)" + +#~ msgid "Stage 2: FlagTree Compiler Integration" +#~ msgstr "阶段 2:FlagTree 编译器接入" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_03_flaggems.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_03_flaggems.mo new file mode 100644 index 0000000000..e57c514207 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_03_flaggems.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_03_flaggems.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_03_flaggems.po new file mode 100644 index 0000000000..e3aa8712f9 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_03_flaggems.po @@ -0,0 +1,147 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:2 +msgid "" +"Stage 3: Operator Adaptation - FlagGems Coverage (Entry-to-Intermediate " +"Progression)" +msgstr "阶段 3:算子适配 — FlagGems 覆盖(初级→中级推进)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:4 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:6 +msgid "The FlagTree branch is available, and the chip Triton backend is stable." +msgstr "FlagTree 分支可用,芯片 Triton 后端稳定" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:8 +msgid "**Hardware investment**" +msgstr "**硬件投入**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:10 +msgid "1 CI machine reused from Stage 2 and 1 optional development/debug machine." +msgstr "CI 机 1 台(复用阶段 2)+ 开发调试机 1 台(可选)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:12 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:14 +msgid "" +"Confirm that the CI machine is connected to the FlagOS CI/CD resource " +"pool." +msgstr "确认 CI 机接入 FlagOS CI/CD 资源池" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:15 +msgid "Run the FlagGems test suite and identify failed or incompatible operators." +msgstr "运行 FlagGems 测试套件,识别失败/不兼容算子" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:16 +#, python-format +msgid "" +"Fix operator compatibility issues, targeting coverage of at least 80% of " +"core operators." +msgstr "修复算子兼容性问题,目标覆盖 ≥80% 核心算子" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:17 +msgid "" +"Use KernelGen to automatically generate missing operators and validate " +"them on the chip." +msgstr "利用 KernelGen 自动生成缺失算子,并针对该芯片验证" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:18 +#, python-format +msgid "" +"Tune operator performance, targeting at least 80% of the vendor native " +"operator performance." +msgstr "算子性能调优,目标达到或超过厂商原生算子性能的 80%" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:19 +msgid "Establish an SBOM scanning mechanism." +msgstr "建立 SBOM 扫描机制" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:21 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:23 +msgid "" +"FlagGems operator compatibility report, test coverage data, and SBOM " +"scanning baseline." +msgstr "FlagGems 算子兼容性报告,测试覆盖率数据,SBOM 扫描基线" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:25 +msgid "**FlagOS contacts**" +msgstr "**FlagOS 方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:27 +msgid "FlagGems operator team and KernelGen platform team." +msgstr "FlagGems 算子团队 + KernelGen 平台团队" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:29 +msgid "**Chip company contacts**" +msgstr "**芯片方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:31 +msgid "Operator development engineers." +msgstr "算子开发工程师" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:33 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:35 +msgid "FlagGems operator test pass rate is at least 80%." +msgstr "FlagGems 算子测试通过率 ≥80%" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:36 +msgid "Core LLM operators pass at 100%." +msgstr "核心 LLM 算子 100% 通过" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:37 +#, python-format +msgid "Operator performance reaches at least 80% of vendor native performance." +msgstr "算子性能 ≥ 厂商原生性能的 80%" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:38 +msgid "The CI test pipeline runs stably for at least two weeks." +msgstr "CI 测试流水线稳定运行 ≥2 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:40 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:42 +msgid "4-8 weeks" +msgstr "4-8 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:44 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:46 +msgid "" +"[FlagGems operator library specification " +"(trial)](https://jwolpxeehx.feishu.cn/docx/GawHdXsuRomaQNxpISec30lxnFg)
[KernelGen" +" Wiki](https://docs.flagos.io/projects/kernelgen/en/latest/)" +msgstr "" +"[FlagGems算子库规范[试行]](https://jwolpxeehx.feishu.cn/docx/GawHdXsuRomaQNxpISec30lxnFg)
[KernelGen" +" Wiki](https://docs.flagos.io/projects/kernelgen/en/latest/)" + +#~ msgid "Stage 3: FlagGems Operator Adaptation" +#~ msgstr "阶段 3:FlagGems 算子适配" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.mo new file mode 100644 index 0000000000..8aa9bf9f78 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.po new file mode 100644 index 0000000000..a357b22609 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.po @@ -0,0 +1,228 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:1 +msgid "" +"Stage 4: Communication and Training/Inference Framework Adaptation " +"(Intermediate Core)" +msgstr "阶段 4:通信与训练推理框架适配(中级核心)" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:3 +msgid "Sub-stage 4a: Communication Adaptation - FlagCX" +msgstr "子阶段 4a:通信适配 — FlagCX" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:5 +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:49 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:7 +msgid "Operator coverage reaches the intermediate standard." +msgstr "算子覆盖率达中级标准" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:9 +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:53 +msgid "**Hardware investment**" +msgstr "**硬件投入**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:11 +msgid "" +"At least two servers, each with at least four cards, connected through " +"high-speed interconnect such as IB or RoCE to form a homogeneous cluster." +msgstr "至少 2 台服务器,每台 ≥4 卡,高速互联(IB/RoCE),构成同构集群" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:13 +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:57 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:15 +msgid "Wrap the vendor native communication library as a unified FlagCX adapter." +msgstr "将厂商原生通信库封装为 FlagCX 统一适配器" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:16 +msgid "" +"Implement core collective communication operations: allreduce, " +"reducescatter, allgather, send, and recv." +msgstr "实现核心集合通信操作:allreduce、reducescatter、allgather、send/recv" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:17 +#, python-format +msgid "" +"Validate homogeneous cluster communication performance, targeting at " +"least 99% of the native library." +msgstr "验证同构集群通信性能,目标 ≥ 原生库 99% 吞吐" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:18 +msgid "" +"Validate heterogeneous cluster mixed communication efficiency, targeting " +"at least 81%." +msgstr "验证异构集群混合通信效率,目标 ≥81%" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:20 +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:65 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:22 +msgid "FlagCX communication adapter and communication performance test report." +msgstr "FlagCX 通信适配器,通信性能测试报告" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:24 +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:69 +msgid "**FlagOS contacts**" +msgstr "**FlagOS 方接口人**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:26 +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:71 +msgid "FlagOS framework R&D team." +msgstr "FlagOS 框架研发团队" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:28 +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:73 +msgid "**Chip company contacts**" +msgstr "**芯片方接口人**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:30 +msgid "Communication library and network engineers." +msgstr "通信库/网络工程师" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:32 +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:77 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:34 +msgid "" +"Collective communication executes correctly in single-node multi-card " +"environments." +msgstr "集合通信在单机多卡环境正确执行" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:35 +#, python-format +msgid "Linearity reaches at least 90% on clusters with eight or more nodes." +msgstr "8 节点以上集群线性度 ≥90%" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:36 +#, python-format +msgid "Homogeneous cluster throughput reaches at least 99% of the native library." +msgstr "同构集群吞吐 ≥原生库 99%" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:37 +msgid "Heterogeneous mixed communication efficiency reaches at least 81%." +msgstr "异构混合效率 ≥81%" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:39 +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:83 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:41 +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:85 +msgid "3-5 weeks" +msgstr "3-5 周" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:43 +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:87 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:45 +msgid "" +"[FlagCX vendor adaptation " +"guide](https://jwolpxeehx.feishu.cn/docx/Vl1NdA8jiohjl6xZV2GcfV6un4d)" +msgstr "" +"[FlagCX " +"厂商适配](https://jwolpxeehx.feishu.cn/docx/Vl1NdA8jiohjl6xZV2GcfV6un4d)" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:47 +msgid "Sub-stage 4b: Training and Inference Adaptation - FlagScale" +msgstr "子阶段 4b:训练推理适配 — FlagScale" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:51 +msgid "FlagCX communication validation has passed." +msgstr "FlagCX 通信验证通过" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:55 +msgid "Reuse the two Stage 4a devices, with persistent storage mounted." +msgstr "复用阶段 4a 的 2 台设备,增加持久化存储挂载" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:59 +msgid "Adapt the Megatron-LM plugin for the training backend." +msgstr "适配 Megatron-LM 插件(训练后端)" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:60 +msgid "Adapt the vLLM/SGLang plugins for the inference backend." +msgstr "适配 vLLM/SGLang 插件(推理后端)" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:61 +msgid "Support TP, PP, DP, EP, and other parallel strategies." +msgstr "支持 TP、PP、DP、EP 等并行策略" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:62 +msgid "Support FP16, BF16, and FP8 mixed-precision training." +msgstr "支持 FP16/BF16/FP8 混合精度训练" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:63 +msgid "Adapt vllm-plugin-FL and provide multi-chip inference services." +msgstr "适配 vllm-plugin-FL,实现多芯片推理服务" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:67 +msgid "FlagScale integration report and training/inference validation results." +msgstr "FlagScale 集成报告,训练/推理验证结果" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:75 +msgid "Framework engineers." +msgstr "框架工程师" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:79 +msgid "A representative model can complete one iteration." +msgstr "代表模型可完整训练一个 iteration" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:80 +msgid "" +"The inference service responds normally end to end, with accuracy " +"deviation below 2%." +msgstr "推理服务端到端响应正常,精度偏差 <2%" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:81 +msgid "The distributed training convergence curve matches the NVIDIA baseline." +msgstr "分布式训练收敛曲线与 NVIDIA 基线一致" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_04_communication_framework.md:89 +msgid "" +"[FlagOS reference " +"material](https://jwolpxeehx.feishu.cn/docx/IiAXdz9ZWouEi0xqMMCcUUU7nJe)[TransformerEngine-FL plugin mechanism user " +"guide](https://jwolpxeehx.feishu.cn/docx/HC0mdLIOFoGv0sxyo8XcFloinOd)[vLLM-plugin-FL plugin mechanism user " +"guide](https://jwolpxeehx.feishu.cn/docx/Oxeyd2WpMo3XTCxqRNMcSCKHngd)[sglang-plugin-FL vendor adaptation development " +"guide](https://jwolpxeehx.feishu.cn/wiki/ReW9w8Kguihzm3kAu8GcSIjsn34)[veRL-FL adaptation " +"documentation](https://jwolpxeehx.feishu.cn/docx/WF7Kdd8ploywmlxBETfc0CWonne)" +msgstr "" +"[FlagOS " +"训练全流程](https://jwolpxeehx.feishu.cn/docx/IiAXdz9ZWouEi0xqMMCcUUU7nJe)[TransformerEngine-FL " +"插件化机制使用手册](https://jwolpxeehx.feishu.cn/docx/HC0mdLIOFoGv0sxyo8XcFloinOd)[vLLM-plugin-FL " +"插件化机制使用手册](https://jwolpxeehx.feishu.cn/docx/Oxeyd2WpMo3XTCxqRNMcSCKHngd)[sglang-plugin-FL " +"厂商适配开发指南](https://jwolpxeehx.feishu.cn/wiki/ReW9w8Kguihzm3kAu8GcSIjsn34)[veRL-FL " +"适配文档](https://jwolpxeehx.feishu.cn/docx/WF7Kdd8ploywmlxBETfc0CWonne)" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_05_validation.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_05_validation.mo new file mode 100644 index 0000000000..c8d14ae4fa Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_05_validation.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_05_validation.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_05_validation.po new file mode 100644 index 0000000000..7bc1aab192 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_05_validation.po @@ -0,0 +1,138 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:2 +msgid "" +"Stage 5: End-to-End Validation and Legal Compliance (Intermediate Closing" +" Gate)" +msgstr "阶段 5:全流程验证与法务合规(中级收尾关卡)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:4 +msgid "" +"For cloud chip adaptation, this page corresponds to Stage 5. This stage " +"follows Stage 4a/4b communication and training/inference framework " +"adaptation. It focuses on validating CI/CD, system validation testing, " +"CTS compatibility certification, and legal compliance." +msgstr "" +"云端芯片适配中,本页对应阶段 5。该阶段承接阶段 4a/4b 的通信与训练推理框架适配结果,重点验证 CI/CD、系统验证测试、CTS " +"兼容性认证和法务合规。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:6 +msgid "**Stage ID**" +msgstr "**阶段标识**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:8 +msgid "Stage 5" +msgstr "阶段 5" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:10 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:12 +msgid "" +"Technical adaptation is complete for Stages 2-4, and the CI/CD resource " +"pool is ready." +msgstr "技术适配完成(阶段 2-4),CI/CD 资源池就绪" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:14 +msgid "**Hardware investment**" +msgstr "**硬件投入**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:16 +msgid "" +"1 CI machine and 2 SVT (system validation testing) machines. SVT machines" +" must be physically isolated from CI machines." +msgstr "CI 机 1 台 + SVT(系统验证测试)机 2 台;SVT 与 CI 机物理隔离" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:18 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:20 +msgid "Validate SIG repository functionality." +msgstr "Sig 仓功能验证" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:21 +msgid "Run DevSecOps automated build, deployment, and validation." +msgstr "DevSecOps 自动化构建、部署、验证" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:22 +msgid "Validate SBOM scanning." +msgstr "SBOM 扫描验证" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:23 +msgid "Run FlagOS CTS compatibility certification tests." +msgstr "运行 FlagOS CTS 兼容性认证测试" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:24 +msgid "Perform legal compliance review." +msgstr "法务合规审查" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:26 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:28 +msgid "" +"Functional validation report, SBOM report, CTS test report, and legal " +"compliance review opinion." +msgstr "功能验证报告、SBOM 报告、CTS 测试报告、法务合规审查意见书" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:30 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:32 +msgid "SIG repository test cases pass at 100%." +msgstr "Sig 仓测试用例 100% 通过" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:33 +msgid "All CTS test items pass or have an explicit waiver list." +msgstr "CTS 测试项全部通过或有明确豁免清单" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:34 +msgid "SBOM scanning reports no critical vulnerabilities." +msgstr "SBOM 扫描无高危漏洞" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:35 +msgid "Legal review has no unresolved items." +msgstr "法务审查无未解决项" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:36 +msgid "" +"CI/CD and SVT end-to-end automation runs for at least seven days without " +"interruption." +msgstr "CI/CD + SVT 全流程自动化运行 ≥7 天无中断" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:38 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:40 +msgid "3-4 weeks" +msgstr "3-4 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:42 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_cloud.md:44 +msgid "" +"SIG repository test specification, CTS specification, SBOM standard, " +"compliance review checklist, and DevSecOps practice guide." +msgstr "Sig 仓测试规范、CTS 规范、SBOM 标准、合规审查清单、DevSecOps 实践指南" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_06_release.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_06_release.mo new file mode 100644 index 0000000000..93875c16ae Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_06_release.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_06_release.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_06_release.po new file mode 100644 index 0000000000..31557d6a31 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_06_release.po @@ -0,0 +1,145 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:2 +msgid "" +"Stage 6: FlagRelease Release and Commercial Promotion (Intermediate-to-" +"Advanced Progression)" +msgstr "阶段 6:FlagRelease 发布与商业推广(中级→高级)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:4 +msgid "" +"For cloud chip adaptation, this page corresponds to Stage 6. This stage " +"starts after End-to-End Validation and Legal Compliance are complete. It " +"focuses on FlagOS certification, FlagRelease image release, model " +"migration, and joint commercial promotion." +msgstr "" +"云端芯片适配中,本页对应阶段 6。该阶段在全流程验证与法务合规完成后启动,重点完成 FlagOS 认证、FlagRelease " +"镜像发布、模型迁移和联合商业推广。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:6 +msgid "**Stage ID**" +msgstr "**阶段标识**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:8 +msgid "Stage 6" +msgstr "阶段 6" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:10 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:12 +msgid "Mainline merge is complete, and certification tests have passed." +msgstr "主干合入完成,认证测试通过" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:14 +msgid "**Hardware investment**" +msgstr "**硬件投入**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:16 +msgid "" +"1 dedicated FlagRelease machine, configured independently and not shared " +"with CI or SVT." +msgstr "FlagRelease 专用机 1 台(独立配置,不与 CI/SVT 混用)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:18 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:20 +msgid "Obtain official FlagOS certification for the device." +msgstr "设备获得 FlagOS 正式认证" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:21 +msgid "Run model migration and image builds on the FlagRelease platform." +msgstr "FlagRelease 平台运行模型迁移和镜像构建" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:22 +msgid "Release Docker images for this chip, expanding the release list over time." +msgstr "发布该芯片的 Docker 镜像(按实际发布清单持续扩展)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:23 +msgid "Join the FlagOS large-model T+0 release plan." +msgstr "加入 FlagOS 大模型 T+0 发布计划" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:24 +msgid "Integrate with partner cloud and MaaS platforms." +msgstr "带入合作云平台和 MaaS 平台" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:25 +msgid "" +"Promote the vendor to intermediate or advanced open-source community " +"membership." +msgstr "上升为开源社区中级/高级会员" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:26 +msgid "Prepare joint branding exposure." +msgstr "联合品牌露出" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:28 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:30 +msgid "" +"FlagOS certification certificate, FlagRelease image list, and joint " +"branding plan." +msgstr "FlagOS 认证证书、FlagRelease 镜像清单、联合品牌方案" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:32 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:34 +msgid "FlagOS compatibility certification is passed." +msgstr "通过 FlagOS 兼容性认证" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:35 +msgid "" +"At least 10 chip-adapted model images are published on the FlagRelease " +"platform." +msgstr "FlagRelease 平台上线 ≥10 个模型的芯片适配镜像" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:36 +msgid "At least one joint customer case is delivered." +msgstr "至少 1 个联合客户案例落地" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:37 +msgid "Dedicated FlagRelease machine availability is at least 99%." +msgstr "FlagRelease 专用机可用性 ≥99%" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:39 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:41 +msgid "4-6 weeks" +msgstr "4-6 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:43 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_06_release_cloud.md:45 +msgid "" +"Certification standard, image release specification, and joint branding " +"guide
[FlagRelease " +"Wiki](https://docs.flagos.io/projects/FlagRelease/en/latest/)" +msgstr "" +"认证标准文档、镜像发布规范、品牌联合指南
[FlagRelease " +"Wiki](https://docs.flagos.io/projects/FlagRelease/en/latest/)" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.mo new file mode 100644 index 0000000000..18c829dfcc Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.po new file mode 100644 index 0000000000..e533211ac8 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.po @@ -0,0 +1,172 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:1 +msgid "" +"Stage 7: KernelGen Adaptation, Automatic Operator Generation, and " +"Operator Competition" +msgstr "阶段 7:KernelGen 适配、算子自动生成与算子大赛" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:3 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:5 +msgid "" +"Mainline merge is complete, FlagOS certification has passed, the " +"FlagRelease image has been released, and the chip compiler, runtime, " +"profiler, and base operator library are available." +msgstr "主干合入完成,FlagOS 认证通过,FlagRelease 镜像已发布,芯片编译器 / Runtime / Profiler / 基础算子库可用" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:7 +msgid "**Hardware investment**" +msgstr "**硬件投入**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:9 +msgid "" +"At least one dedicated KernelGen machine. Multi-card test environments " +"are required for multi-card or multi-node optimization. Operator " +"competitions require a separate evaluation machine or isolated queue." +msgstr "KernelGen 专用机 ≥1 台;如支持多卡、多节点优化,需提供多卡测试环境;算子大赛需提供独立评测机或隔离队列" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:11 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:13 +msgid "Connect to the KernelGen automatic operator generation platform." +msgstr "接入 KernelGen 自动算子生成平台" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:14 +msgid "" +"Configure compilation, runtime, validation, and benchmark environments " +"for the chip." +msgstr "配置该芯片的编译、运行、验证、Benchmark 环境" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:15 +msgid "Connect the chip profiler for automatic performance bottleneck analysis." +msgstr "接入芯片 Profiler,实现性能瓶颈自动分析" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:16 +msgid "" +"Establish performance baselines for PyTorch, Triton, TLE, and native " +"libraries." +msgstr "建立 PyTorch / Triton / TLE / 原生库性能基线" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:17 +msgid "" +"Prioritize high-frequency LLM operators, including Attention, MoE, " +"RMSNorm, RoPE, TopK, KV Cache, communication fusion, and others." +msgstr "优先优化 LLM 高频算子,包括 Attention、MoE、RMSNorm、RoPE、TopK、KV Cache、通信融合等" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:18 +msgid "Automatically generate operators and submit PRs to FlagGems and FlagOS." +msgstr "自动生成算子并向 FlagGems / FlagOS 提交 PR" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:19 +msgid "Build the chip's KernelGen Skills and optimization knowledge base." +msgstr "沉淀该芯片的 KernelGen Skills 和优化知识库" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:20 +msgid "Connect to the KernelGenBench multi-chip evaluation leaderboard." +msgstr "接入 KernelGenBench 多芯片评测榜单" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:21 +msgid "Co-host an operator optimization competition or KernelGen Hackathon." +msgstr "联合举办算子优化赛 / KernelGen Hackathon" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:22 +msgid "" +"Include outstanding operators in the FlagRelease image and large-model " +"T+0 release process." +msgstr "将优秀算子纳入 FlagRelease 镜像和大模型 T+0 发布流程" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:24 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:26 +msgid "" +"KernelGen chip adaptation report, operator generation environment, " +"profiler integration documentation, performance baseline report, " +"generated-operator PR list, merged-operator list, KernelGen Skills, " +"KernelGenBench results, operator competition plan, and leaderboard." +msgstr "" +"KernelGen 芯片适配报告、算子生成环境、Profiler 接入文档、性能基线报告、生成算子 PR 清单、合入算子清单、KernelGen " +"Skills、KernelGenBench 结果、算子大赛方案与榜单" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:28 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:30 +msgid "" +"Complete the automatic generation, compilation, correctness validation, " +"benchmark, and PR loop." +msgstr "完成自动生成 → 编译 → 正确性验证 → Benchmark → PR 的闭环" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:31 +msgid "Cover at least 30 high-frequency operators in the first batch." +msgstr "首批覆盖 ≥30 个高频算子" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:32 +msgid "Produce at least 10 mergeable PRs." +msgstr "形成 ≥10 个可合入 PR" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:33 +msgid "Merge at least 5 FlagGems or FlagOS operators." +msgstr "合入 ≥5 个 FlagGems / FlagOS 算子" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:34 +msgid "" +"Achieve at least 1.1x average speedup for key operators with no " +"performance regression." +msgstr "重点算子平均加速 ≥1.1x 且无性能回退" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:35 +msgid "Document at least 20 chip optimization Skills." +msgstr "沉淀 ≥20 条芯片优化 Skill" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:36 +msgid "Cover at least one official KernelGenBench track." +msgstr "KernelGenBench 至少覆盖 1 个正式赛道" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:37 +msgid "Deliver at least one joint technical case study." +msgstr "至少 1 个联合技术案例落地" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:39 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:41 +msgid "6-10 weeks" +msgstr "6-10 周" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:43 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_07_kernelgen.md:45 +msgid "" +"KernelGen chip integration specification, operator generation " +"specification, FlagGems PR specification, KernelGenBench evaluation " +"specification, profiler integration guide, and operator competition " +"rules." +msgstr "" +"KernelGen 芯片接入规范、算子生成规范、FlagGems PR 规范、KernelGenBench 评测规范、Profiler " +"接入指南,算子大赛规则文档" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.mo new file mode 100644 index 0000000000..a08b1710bd Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.po new file mode 100644 index 0000000000..a404bc08e5 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.po @@ -0,0 +1,55 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.md:1 +msgid "Stage 8: Image Build" +msgstr "阶段 8:Image 构建" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.md:3 +msgid "" +"Image builds are divided into three levels: base image, runtime image, " +"and application image." +msgstr "镜像的构建分为三个层级:base image、runtime image、application image。" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.md:5 +msgid "Follow these documentation guides:" +msgstr "可遵循以下文档指引:" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.md:7 +msgid "" +"[Release Info Overview](https://flagos-ai.github.io/release-" +"info/overview/)" +msgstr "[Release Info 概览](https://flagos-ai.github.io/release-info/overview/)" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.md:8 +msgid "" +"[Release Info Onboarding Guide](https://flagos-ai.github.io/release-" +"info/contribution/onboarding/)" +msgstr "" +"[Release Info Onboarding 指南](https://flagos-ai.github.io/release-" +"info/contribution/onboarding/)" + +#: ../../chip_adaptation_guide/cloud/04_adaptation_plan/stage_08_image_build.md:10 +msgid "" +"If you would like to join, submit an issue in the [release-info " +"repository](https://github.com/flagos-ai/release-info/issues)." +msgstr "" +"如有意向加入,可在 [release-info 仓库提交 issue](https://github.com/flagos-ai/release-" +"info/issues)。" + +#~ msgid "Image Build (Stage 8)" +#~ msgstr "Image 构建(阶段 8)" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_architecture.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_architecture.mo new file mode 100644 index 0000000000..d6d0f72eba Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_architecture.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_architecture.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_architecture.po new file mode 100644 index 0000000000..1411d2b5f4 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_architecture.po @@ -0,0 +1,32 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_architecture.md:2 +msgid "Certification Architecture" +msgstr "认证架构" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_architecture.md:4 +msgid "![image5.png](/chip_adaptation_guide/assets/common/certification_architecture-en.png)" +msgstr "![image5.png](/chip_adaptation_guide/assets/common/certification_architecture.png)" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_architecture.md:4 +msgid "image5.png" +msgstr "image5.png" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_architecture.md:6 +msgid "Figure 4. FlagOS certification architecture" +msgstr "图 4 FlagOS 认证架构" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.mo new file mode 100644 index 0000000000..ab4c29939a Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.po new file mode 100644 index 0000000000..c4f86ea764 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.po @@ -0,0 +1,101 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:38+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_levels.md:2 +msgid "Certification Levels" +msgstr "四级认证等级" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "**Certification level**" +msgstr "**认证等级**" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "**Requirements**" +msgstr "**要求**" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "**Applicable objects**" +msgstr "**适用对象**" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "**Grant conditions**" +msgstr "**授予条件**" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "FlagOS Ready" +msgstr "FlagOS Ready" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "Basic compatibility certification" +msgstr "基础兼容认证" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "All chips entering intermediate cooperation" +msgstr "进入中级的所有芯片" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "Core CTS test items pass" +msgstr "CTS 核心测试项通过" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "FlagOS Verified" +msgstr "FlagOS Verified" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "Functionality, performance, and security" +msgstr "功能 + 性能 + 安全" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +# +msgid "Chips that complete intermediate adaptation" +msgstr "完成中级适配的芯片" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "CTS, VTS, and STS all pass" +msgstr "CTS + VTS + STS 全部通过" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "FlagOS Interoperable" +msgstr "FlagOS Interoperable" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "Heterogeneous interoperability certification" +msgstr "异构互操作认证" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "Chips entering advanced cooperation" +msgstr "进入高级的芯片" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "CTS, VTS, STS, and ITS all pass" +msgstr "CTS + VTS + STS + ITS 全部通过" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "FlagOS Platinum" +msgstr "FlagOS Platinum" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +# +msgid "Flagship certification" +msgstr "旗舰级认证" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "T+0 release partners" +msgstr "T+0 发布合作伙伴" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_levels.md:1 +msgid "All tests pass and the partner contributes to the community" +msgstr "全部测试通过 + 社区共建贡献" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_maintenance.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_maintenance.mo new file mode 100644 index 0000000000..9dc6d8309f Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_maintenance.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_maintenance.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_maintenance.po new file mode 100644 index 0000000000..0672b02955 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_maintenance.po @@ -0,0 +1,36 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-20 00:00+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_maintenance.md:4 +msgid "Certification Maintenance" +msgstr "认证维护" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_maintenance.md:6 +msgid "Validity period: 12 months; recertification is required at expiration." +msgstr "有效期:12 个月,到期重新认证。" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_maintenance.md:8 +msgid "Changes triggering retesting: major driver upgrades, Triton backend refactoring, or major PyTorch upgrades." +msgstr "变更触发重测:驱动大版本升级、Triton 后端重构、PyTorch 大版本升级。" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_maintenance.md:10 +#, python-format +msgid "Certification revocation: a CI pass rate below 80% for three consecutive months, or an unresolved critical security vulnerability." +msgstr "认证撤销机制:连续 3 个月 CI 通过率 <80%,或出现未修复的高危安全漏洞。" + +msgid "Waiver management: each waiver must state the technical reason, remediation plan, and expected remediation date; the maximum waiver period is six months." +msgstr "豁免管理:每项豁免需明确技术原因、修复计划、预计修复日期,最长豁免期 6 个月。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.mo new file mode 100644 index 0000000000..e3665e07c9 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.po new file mode 100644 index 0000000000..01e178415f --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.po @@ -0,0 +1,90 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:07+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_objects.md:2 +msgid "Three-Tier Certification Objects" +msgstr "三层认证对象" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "**Certification tier**" +msgstr "**认证层级**" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "**Certification object**" +msgstr "**认证对象**" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "**Scope**" +msgstr "**覆盖范围**" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "**Benefits and listing**" +msgstr "**权益/展示**" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "Chip certification" +msgstr "芯片认证" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "Chip and base software stack" +msgstr "芯片和基础软件栈" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "" +"Drivers, SDK, compiler, operators, communication, frameworks, " +"performance, and security" +msgstr "驱动、SDK、编译器、算子、通信、框架、性能和安全" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "Chip compatibility certification, adaptation list, and Day 0 candidate" +msgstr "芯片兼容性认证、适配清单、Day 0 候选" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "System certification" +msgstr "整机认证" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "Complete server or device" +msgstr "服务器/设备整机" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "" +"Multi-card interconnect, thermal management, long-duration stability, " +"SVT, resource monitoring, and system stability" +msgstr "多卡互联、散热、长稳、SVT、资源监控、稳定性" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "System certification and showroom candidate" +msgstr "整机认证、展厅展示候选" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "Software/cloud-native certification" +msgstr "软件/云原生认证" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "OS distributions, containers, Kubernetes, and cloud-native platforms" +msgstr "OS 发行版、容器、Kubernetes、云原生平台" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "" +"Images, scheduling, Kubernetes, system integration, security, and " +"interoperability" +msgstr "镜像、调度、K8s、系统集成、安全和互操作" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/certification_objects.md:1 +msgid "Distribution plan and cloud-native ecosystem certification" +msgstr "发行版计划、云原生生态认证" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_process.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_process.mo new file mode 100644 index 0000000000..3846ab2a76 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_process.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_process.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_process.po new file mode 100644 index 0000000000..a05b8d56ae --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/certification_process.po @@ -0,0 +1,32 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_process.md:2 +msgid "Certification Process" +msgstr "认证流程" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_process.md:4 +msgid "![image6.png](/chip_adaptation_guide/assets/common/certification_process-en.jpeg)" +msgstr "![image6.png](/chip_adaptation_guide/assets/common/certification_process.png)" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_process.md:4 +msgid "image6.png" +msgstr "image6.png" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_process.md:6 +msgid "Figure 5. Certification process" +msgstr "图 5 认证流程" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/index.mo new file mode 100644 index 0000000000..41659aad39 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/index.po new file mode 100644 index 0000000000..5f46e29b66 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/index.po @@ -0,0 +1,23 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-20 00:00+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/index.md:4 +msgid "FlagOS Compatibility Certification System" +msgstr "FlagOS 兼容性认证体系" + +msgid "Following the Android CDD/CTS design, FlagOS has established a tiered certification system. Based on further discussion, certification objects are divided into three layers: chips, complete systems, and software/cloud-native platforms, covering different vendor forms and later showroom, policy, and procurement scenarios." +msgstr "参考 Android CDD/CTS 设计,FlagOS 建立分级认证体系;同时结合会议讨论,认证对象进一步分为芯片、整机、软件/云原生三层,以覆盖不同厂商形态和后续展示、政策、集采场景。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.mo new file mode 100644 index 0000000000..de6122a4a2 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.po new file mode 100644 index 0000000000..7012909bfd --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.po @@ -0,0 +1,263 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:07+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_cloud.md:2 +msgid "Test Suites" +msgstr "测试套件详情" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_cloud.md:4 +msgid "CTS (Compatibility Test Suite) - Compatibility Test Suite" +msgstr "CTS(Compatibility Test Suite)— 兼容性测试套件" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "**Test domain**" +msgstr "**测试域**" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "**Coverage**" +msgstr "**覆盖内容**" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "**Pass standard**" +msgstr "**通过标准**" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "Compiler compatibility" +msgstr "编译器兼容性" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "FlagTree compilation and FLIR IR mapping correctness" +msgstr "FlagTree 编译 + FLIR IR 映射正确性" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "100%" +msgstr "100%" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "Operator compatibility" +msgstr "算子兼容性" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "FlagGems core operator correctness" +msgstr "FlagGems 核心算子功能正确性" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid ">=95%" +msgstr "≥95%" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "Communication compatibility" +msgstr "通信兼容性" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "FlagCX collective communication correctness" +msgstr "FlagCX 集合通信功能正确性" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "Framework compatibility" +msgstr "框架兼容性" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "Basic FlagScale training and inference workflow" +msgstr "FlagScale 训练/推理基本流程" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "PyTorch API compatibility" +msgstr "PyTorch API 兼容性" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "Correctness of common aten API return values" +msgstr "常用 aten API 返回值正确性" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid ">=90%" +msgstr "≥90%" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "Triton API compatibility" +msgstr "Triton API 兼容性" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "Triton language feature support" +msgstr "Triton 语言特性支持度" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "Accuracy consistency" +msgstr "精度一致性" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "Comparison with the NVIDIA baseline" +msgstr "与 NVIDIA 基线对比,关键指标" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:5 +msgid "Deviation <2%" +msgstr "偏差 <2%" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_common.md:2 +msgid "VTS (Verification Test Suite) - Verification Test Suite" +msgstr "VTS(Verification Test Suite)— 验证测试套件" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Stress test" +msgstr "压力测试" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Continuous operator execution for 72 hours" +msgstr "连续 72 小时算子执行" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "No crashes" +msgstr "无崩溃" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Stability test" +msgstr "稳定性测试" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "1,000 repeated training and inference runs" +msgstr "重复 1000 次训练/推理流程" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Consistent results" +msgstr "结果一致" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Long-duration stability" +msgstr "长稳测试" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Continuous 24/7 operation" +msgstr "7×24 持续运行" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "No memory leaks" +msgstr "无内存泄漏" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Resource monitoring" +msgstr "资源监控" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "GPU utilization, memory bandwidth, and temperature" +msgstr "GPU 利用率、内存带宽、温度" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Within vendor specifications" +msgstr "在厂商规格范围内" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_cloud.md:19 +msgid "STS (Security Test Suite) - Security Test Suite" +msgstr "STS(Security Test Suite)— 安全测试套件" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "SBOM scanning" +msgstr "SBOM 扫描" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Dependency vulnerability scanning" +msgstr "依赖组件漏洞扫描" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "No high-risk vulnerabilities or CVEs" +msgstr "无高危/CVE" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Driver security" +msgstr "驱动安全" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Driver file permission and signature validation" +msgstr "驱动文件权限、签名验证" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Passed" +msgstr "通过" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Container security" +msgstr "容器安全" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Container image scanning" +msgstr "镜像扫描" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "No high-risk vulnerabilities" +msgstr "无高危" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Communication encryption" +msgstr "通信加密" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "FlagCX communication-link encryption validation" +msgstr "FlagCX 通信链路加密验证" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_cloud.md:28 +msgid "ITS (Interoperability Test Suite) - Interoperability Test Suite" +msgstr "ITS(Interoperability Test Suite)— 互操作性测试套件" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Heterogeneous mixed training" +msgstr "异构混合训练" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Chip and NVIDIA mixed training" +msgstr "该芯片 + NVIDIA 混合训练" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Efficiency >=81%" +msgstr "效率 ≥81%" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Heterogeneous mixed inference" +msgstr "异构混合推理" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Chip and NVIDIA mixed inference" +msgstr "该芯片 + NVIDIA 混合推理" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Accuracy deviation <2%" +msgstr "精度偏差 <2%" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Cross-vendor communication" +msgstr "跨厂商通信" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "FlagCX cross-vendor testing" +msgstr "FlagCX 跨厂商测试" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Correctness 100%" +msgstr "正确性 100%" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "Multi-framework compatibility" +msgstr "多框架兼容" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "FlagScale, vLLM, and SGLang" +msgstr "FlagScale + vLLM + SGLang" + +#: ../../chip_adaptation_guide/cloud/05_compatibility_certification/test_suites.md:16 +msgid "All passed" +msgstr "全部通过" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/ai_distribution.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/ai_distribution.mo new file mode 100644 index 0000000000..51f8f50518 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/ai_distribution.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/ai_distribution.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/ai_distribution.po new file mode 100644 index 0000000000..9db8c63f00 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/ai_distribution.po @@ -0,0 +1,84 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 15:50+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:2 +msgid "AI Distribution Validation" +msgstr "AI 发行版验证" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:4 +msgid "**Validation party**" +msgstr "**验证方**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:6 +msgid "AI distribution company" +msgstr "AI 发行版公司" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:8 +msgid "**Validation scope**" +msgstr "**验证范围**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:10 +msgid "Complete installation and operation of PyTorch on the chip." +msgstr "PyTorch 在该芯片上的完整安装和运行" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:11 +msgid "Loading and execution of the FlagGems operator library." +msgstr "FlagGems 算子库加载和执行" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:12 +msgid "End-to-end operation of the Triton compilation pipeline." +msgstr "Triton 编译流水线端到端工作" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:13 +msgid "Loading and inference of representative models such as Qwen2.5 and Llama3." +msgstr "代表模型(如 Qwen2.5、Llama3)加载和推理" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:15 +msgid "**Hardware requirements**" +msgstr "**硬件要求**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:17 +msgid "The dedicated machine provided by the chip company during the FlagRelease stage can be made available for remote use by the AI distribution." +msgstr "芯片公司在 FlagRelease 阶段提供的专用机可开放给 AI 发行版远程使用" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:19 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:21 +msgid "PyTorch core test cases pass." +msgstr "PyTorch 核心测试用例通过" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:22 +msgid "The FlagGems operator loading rate reaches 100%." +msgstr "FlagGems 算子加载率 100%" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:23 +msgid "Triton kernels compile and execute correctly." +msgstr "Triton kernel 编译和执行正确" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:25 +msgid "**Relationship to certification**" +msgstr "**与认证的关系**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:27 +msgid "AI distribution validation is an important reference condition for FlagOS Platinum certification." +msgstr "AI 发行版验证是 FlagOS Platinum 认证的重要参考条件" + + + + + diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.mo new file mode 100644 index 0000000000..1c65b1a41d Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.po new file mode 100644 index 0000000000..c60bb4606a --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.po @@ -0,0 +1,80 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:07+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/compatibility_declaration.md:2 +msgid "Compatibility Declaration" +msgstr "发行版兼容性声明" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "**Declaration level**" +msgstr "**声明等级**" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "**Meaning**" +msgstr "**含义**" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "**Condition**" +msgstr "**条件**" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "Platinum Partner" +msgstr "Platinum Partner" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "The chip is fully supported in the distribution." +msgstr "该芯片在发行版中完全受支持" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "Passes Level 1 + Level 2 + Level 3 validation and is an advanced partner." +msgstr "通过层级 1 + 2 + 3 验证,且为高级合作伙伴" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "Certified" +msgstr "Certified" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "The chip is certified to run normally in the distribution." +msgstr "该芯片在发行版中经认证可正常运行" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "Passes Level 1 + Level 2 validation." +msgstr "通过层级 1 + 2 验证" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "Compatible" +msgstr "Compatible" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "The chip is verified as compatible in the distribution." +msgstr "该芯片在发行版中经验证兼容" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "Passes at least Level 1 validation." +msgstr "至少通过层级 1 验证" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "Community Support" +msgstr "Community Support" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "Community-driven support without official distribution validation." +msgstr "社区驱动,发行版未做官方验证" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/compatibility_declaration.md:1 +msgid "Only FlagOS CTS has passed." +msgstr "仅通过 FlagOS CTS" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/index.mo new file mode 100644 index 0000000000..cb44129776 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/index.po new file mode 100644 index 0000000000..0044bff16e --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/index.po @@ -0,0 +1,23 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-20 00:00+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/index.md:4 +msgid "Distribution Validation System (Draft)" +msgstr "发行版公司验证体系(草稿)" + +msgid "End users obtain FlagOS capabilities through operating-system and AI distributions. Distribution companies must validate chip compatibility in their distributions so that adaptation results move beyond experimental environments into real delivery pipelines." +msgstr "芯片适配的最终用户通过操作系统发行版和 AI 发行版获取 FlagOS 能力。发行版公司需要在各自发行版中验证芯片兼容性,确保适配成果不仅停留在实验环境,而是进入真实交付链路。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.mo new file mode 100644 index 0000000000..b122767066 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.po new file mode 100644 index 0000000000..d1a2cb1dc6 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.po @@ -0,0 +1,106 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:07+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/joint_validation.md:2 +msgid "Joint Validation Process" +msgstr "发行版联合验证流程" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "**Step**" +msgstr "**步骤**" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "**Action**" +msgstr "**动作**" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "**Result**" +msgstr "**结果**" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "1" +msgstr "1" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "The chip passes FlagOS CTS" +msgstr "芯片通过 FlagOS CTS" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "CTS report and certification level" +msgstr "形成 CTS 报告和认证级别" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "2" +msgstr "2" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "" +"FlagOS sends the chip certification information to the distribution " +"company" +msgstr "FlagOS 向发行版公司推送芯片认证信息" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "Certification level, CTS report, and device access method" +msgstr "包含认证级别、CTS 报告、设备接入方式" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "3" +msgstr "3" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "OS distribution validation" +msgstr "OS 发行版验证" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "OS distribution validation report" +msgstr "生成 OS 发行版验证报告" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "4" +msgstr "4" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "AI distribution validation" +msgstr "AI 发行版验证" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "AI distribution validation report" +msgstr "生成 AI 发行版验证报告" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "5" +msgstr "5" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "ISV validation (optional)" +msgstr "ISV 验证(可选)" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "Third-party compatibility declaration" +msgstr "生成第三方兼容性声明" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "6" +msgstr "6" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "The distribution lists the chip" +msgstr "发行版收录芯片" + +#: ../../chip_adaptation_guide/cloud/06_distribution_validation/joint_validation.md:1 +msgid "Entry in the \"FlagOS Compatible\" list" +msgstr "进入 “FlagOS Compatible” 列表成员" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/os_distribution.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/os_distribution.mo new file mode 100644 index 0000000000..75e38eb8ff Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/os_distribution.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/os_distribution.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/os_distribution.po new file mode 100644 index 0000000000..fedf2748cc --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/os_distribution.po @@ -0,0 +1,84 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 15:50+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:2 +msgid "OS Distribution Validation" +msgstr "OS 发行版验证" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:4 +msgid "**Validation party**" +msgstr "**验证方**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:6 +msgid "OS distribution company" +msgstr "操作系统发行版公司" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:8 +msgid "**Validation scope**" +msgstr "**验证范围**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:10 +msgid "Driver loading and stability on the target OS kernel version." +msgstr "芯片驱动在目标 OS 内核版本上的加载和稳定性" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:11 +msgid "Compatibility of core system libraries such as glibc, libstdc++, and OpenMPI." +msgstr "基础系统库(glibc、libstdc++、OpenMPI 等)兼容性" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:12 +msgid "Docker/containerd runtime operation." +msgstr "Docker/containerd 容器运行时正常工作" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:13 +msgid "Kernel module signature validation." +msgstr "内核模块签名验证" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:15 +msgid "**Hardware requirements**" +msgstr "**硬件要求**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:17 +msgid "One device provided by the chip company or made available for remote access." +msgstr "芯片公司提供或开放远程访问的 1 台设备" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:19 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:21 +msgid "The driver loads normally on the target kernel, with no errors in dmesg." +msgstr "驱动在目标内核上加载无异常(dmesg 无 error)" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:22 +msgid "Core LTP test items pass." +msgstr "LTP 核心测试项通过" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:23 +msgid "The container runtime can pull and run standard images." +msgstr "容器运行时可拉取并运行标准镜像" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:25 +msgid "**Relationship to certification**" +msgstr "**与认证的关系**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:27 +msgid "After the chip obtains FlagOS Verified certification, it enters the OS distribution validation pipeline." +msgstr "芯片获得 FlagOS Verified 认证后,进入 OS 发行版验证管道" + + + + + diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/validation_levels.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/validation_levels.mo new file mode 100644 index 0000000000..094cfac4e5 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/validation_levels.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/validation_levels.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/validation_levels.po new file mode 100644 index 0000000000..c18e13ba9a --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/06_distribution_validation/validation_levels.po @@ -0,0 +1,32 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/validation_levels.md:2 +msgid "Distribution Validation Levels" +msgstr "发行版验证层级" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/validation_levels.md:4 +msgid "![image7.png](/chip_adaptation_guide/assets/common/distribution_validation-en.png)" +msgstr "![image7.png](/chip_adaptation_guide/assets/common/distribution_validation.png)" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/validation_levels.md:4 +msgid "image7.png" +msgstr "image7.png" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/validation_levels.md:6 +msgid "Figure 6. Distribution validation levels" +msgstr "图 6 发行版验证层级" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/07_quality_metrics/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/07_quality_metrics/index.mo new file mode 100644 index 0000000000..f2cd8ecc0e Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/07_quality_metrics/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/07_quality_metrics/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/07_quality_metrics/index.po new file mode 100644 index 0000000000..038b031eb1 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/07_quality_metrics/index.po @@ -0,0 +1,198 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:38+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/07_quality_metrics/index.md:2 +msgid "Key Quality Metrics" +msgstr "关键检验指标" + +#: ../../chip_adaptation_guide/_shared/07_quality_metrics/cloud.md:2 +# +msgid "" +"**Performance-related metrics must remain key acceptance criteria " +"throughout Stages 3-6.** Stage 3 focuses on operator performance, Stage " +"4a on communication throughput and linearity, and Stage 4b on " +"training/inference accuracy and end-to-end performance. Before advanced " +"cooperation, Day 0, distribution, or showroom plans, the efficiency " +"target relative to the NVIDIA baseline under equivalent compute must be " +"confirmed. The exact percentage will be defined later according to the " +"KT2 requirements." +msgstr "" +"**性能相关指标需作为关键检验项贯穿阶段 3-6:** 阶段 3 关注算子性能,阶段 4a 关注通信吞吐与线性度,阶段 4b " +"关注训练/推理精度与端到端表现;进入高级合作、Day 0、发行版或展厅展示计划前,应额外确认同等算力下相对 NVIDIA 基线的效率指标(X%,待" +" KT2期 口径确定)。" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "**Stage**" +msgstr "**阶段**" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "**Acceptance criteria**" +msgstr "**核心检验指标**" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "**Pass standard**" +msgstr "**通过标准**" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "**Required hardware**" +msgstr "**涉及硬件**" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Stage 0" +msgstr "阶段 0" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "MOU signing" +msgstr "MOU 签署" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Effective after both parties sign or seal" +msgstr "双方盖章生效" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "N/A" +msgstr "—" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Stage 1" +msgstr "阶段 1" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Environment readiness" +msgstr "环境就绪" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Chip is recognized by the OS and PyTorch can recognize the backend" +msgstr "芯片被 OS + PyTorch 识别" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "1 development board" +msgstr "开发板 1 台" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Stage 2" +msgstr "阶段 2" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Triton kernel compilation pass rate" +msgstr "Triton kernel 编译通过率" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid ">=95%, CI machine availability >=99%" +msgstr "≥95%,CI 机器可用性 ≥99%" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "1 CI machine" +msgstr "CI 1 台" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Stage 3" +msgstr "阶段 3" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "FlagGems operator test pass rate" +msgstr "FlagGems 算子测试通过率" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid ">=80%, core operators 100%" +msgstr "≥80%,核心算子 100%" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "1 CI machine and 1 debug machine" +msgstr "CI 1 台 + 调试机 1 台" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Stage 4a" +msgstr "阶段 4a" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Homogeneous collective communication throughput ratio" +msgstr "集合通信同构吞吐比" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +#, python-format +msgid ">=99% of the native library" +msgstr "≥原生库 99%" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "2-machine multi-card cluster" +msgstr "2 台多卡集群" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Stage 4b" +msgstr "阶段 4b" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Model accuracy deviation" +msgstr "模型精度偏差" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +#, python-format +msgid "<2% relative to the NVIDIA baseline" +msgstr "<2%(相对 NVIDIA 基线)" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Reuse Stage 4a devices" +msgstr "复用 4a 设备" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Stage 5" +msgstr "阶段 5" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "CTS core tests passed" +msgstr "CTS 核心测试项通过" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "100% passed or waiver approved" +msgstr "100% 或豁免获批" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "1 CI machine and 2 SVT machines" +msgstr "CI 1 台 + SVT 2 台" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Stage 6" +msgstr "阶段 6" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Number of models released through FlagRelease" +msgstr "FlagRelease 上线模型数" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid ">=10 models" +msgstr "≥10 个" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "1 dedicated FR machine" +msgstr "FR 专机 1 台" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Advanced cooperation" +msgstr "高级合作" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Equivalent-compute performance comparison" +msgstr "同等算力性能对标" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +#, python-format +msgid "Efficiency reaches X% of the NVIDIA equivalent-compute baseline" +msgstr "运行效率达到 NVIDIA 同等算力基线 X%" + +#: ../../chip_adaptation_guide/cloud/07_quality_metrics/index.md:8 +msgid "Multi-card cluster, FR machine, or CI" +msgstr "多卡集群 / FR 专机 / CI" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.mo new file mode 100644 index 0000000000..51130065b4 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.po new file mode 100644 index 0000000000..452c4c1da9 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.po @@ -0,0 +1,125 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:07+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/08_tools_and_skills/skill_catalog.md:2 +msgid "Supporting Skills and Automation Tools" +msgstr "配套 Skill / 自动化工具支持" + +#: ../../chip_adaptation_guide/_shared/08_tools_and_skills/skill_catalog.md:4 +msgid "" +"FlagOS packages multiple adaptation workflow steps as reusable Agent " +"Skills. Chip companies and their development teams can use them according" +" to the adaptation stage." +msgstr "FlagOS 将适配流程的多个步骤封装为可复用的 Agent Skill,芯片公司及其开发团队可按阶段使用。" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "**Skill name**" +msgstr "**Skill 名称**" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "**Applicable stage**" +msgstr "**适用阶段**" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "**Function**" +msgstr "**功能**" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "gpu-container-setup-flagos" +msgstr "gpu-container-setup-flagos" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "Stage 1" +msgstr "阶段 1" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "" +"Automatically detect the GPU vendor, launch a PyTorch container, and " +"validate the GPU/accelerator card." +msgstr "自动检测 GPU 厂商,拉起 PyTorch 容器,验证 GPU/加速卡" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "install-stack-flagos" +msgstr "install-stack-flagos" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "Stages 1-2" +msgstr "阶段 1-2" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "Install the full stack: FlagTree, FlagGems, FlagCX, and FlagScale." +msgstr "全栈安装(FlagTree + FlagGems + FlagCX + FlagScale)" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "kernelgen-flagos" +msgstr "kernelgen-flagos" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "Stage 3" +msgstr "阶段 3" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "Automatically generate and iteratively optimize operators." +msgstr "算子自动生成与迭代优化" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "model-migrate-flagos" +msgstr "model-migrate-flagos" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "Stage 4b" +msgstr "阶段 4b" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "Backport models from upstream vLLM to vllm-plugin-FL." +msgstr "从上游 vLLM 回移植模型到 vllm-plugin-FL" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "model-verify-flagos" +msgstr "model-verify-flagos" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "Stage 5" +msgstr "阶段 5" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "Verify the serving stack step by step." +msgstr "分步验证 serving stack" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "perf-test-flagos" +msgstr "perf-test-flagos" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "Stages 4-5" +msgstr "阶段 4-5" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "Run performance benchmarks." +msgstr "性能基准测试" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "flagrelease-entrance-flagos" +msgstr "flagrelease-entrance-flagos" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "Stage 6" +msgstr "阶段 6" + +#: ../../chip_adaptation_guide/cloud/08_tools_and_skills/skill_catalog.md:1 +msgid "Orchestrate the complete release pipeline." +msgstr "完整发布 pipeline 编排" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/advanced_quality_requirements.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/advanced_quality_requirements.mo new file mode 100644 index 0000000000..ecf9109ccd Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/advanced_quality_requirements.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/advanced_quality_requirements.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/advanced_quality_requirements.po new file mode 100644 index 0000000000..2d5e735cea --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/advanced_quality_requirements.po @@ -0,0 +1,44 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:38+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/advanced_quality_requirements.md:2 +msgid "Advanced Cooperation: Deterministic Performance and Quality Requirements" +msgstr "高级合作的性能与质量确定性要求" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/advanced_quality_requirements.md:4 +msgid "" +"Performance metrics are prerequisites for advanced cooperation, Day 0 " +"plans, distribution validation, and showroom display; they must not be " +"treated as post-release promotional items." +msgstr "性能指标是进入高级合作、Day 0 计划、发行版验证和展厅展示的前置条件,不作为事后宣传项处理。" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/advanced_quality_requirements.md:6 +msgid "" +"Vendors must maintain the mainstream diff so that chip adaptation on the " +"mainline has continuous ownership." +msgstr "厂商承诺维护 Mainstream Diff,确保主线上的芯片适配有人持续跟进。" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/advanced_quality_requirements.md:8 +msgid "" +"FlagOS upgrades and vendor Driver/SDK upgrades must both enter CI/CD " +"validation." +msgstr "FlagOS 升级与厂商 Driver/SDK 升级共同进入 CI/CD 验证。" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/advanced_quality_requirements.md:10 +msgid "" +"Day 0 releases, distribution plans, and policy recommendation plans must " +"meet maturity, support, performance, and quality requirements." +msgstr "Day 0 发布、发行版或政策推荐计划时,需满足成熟度、支持力度、性能和质量要求。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/index.mo new file mode 100644 index 0000000000..d1de68f319 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/index.po new file mode 100644 index 0000000000..c7d5104b4f --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/index.po @@ -0,0 +1,19 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-20 00:00+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +msgid "Promotion Criteria Checklist" +msgstr "晋升条件清单" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_advanced.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_advanced.mo new file mode 100644 index 0000000000..99386f9dcd Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_advanced.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_advanced.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_advanced.po new file mode 100644 index 0000000000..3528ff87e6 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_advanced.po @@ -0,0 +1,59 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:38+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:2 +msgid "Promotion from Intermediate to Advanced" +msgstr "中级 → 高级的必要条件" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:4 +msgid "" +"If the hardware supports FlagCX, complete the adaptation and maintain it " +"continuously; the SVT two-machine cluster must remain stable." +msgstr "如果为支持 FlagCX 的硬件,需适配完成且持续维护,SVT 双机集群稳定" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:6 +msgid "Use the FlagTree compiler by default." +msgstr "默认启用 FlagTree 编译器" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:8 +msgid "Pass the FlagOS CTS, VTS, and STS certification tests." +msgstr "通过 FlagOS CTS + VTS + STS 认证测试" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:10 +msgid "" +"Establish a compatibility matrix for core dependencies such as PyTorch " +"and Triton." +msgstr "建立核心依赖(PyTorch、Triton)兼容性矩阵" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:12 +msgid "" +"Complete the DevSecOps system and close the loop for security incident " +"response." +msgstr "完成 DevSecOps 体系,实现安全事件响应闭环" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:14 +msgid "Co-build the FlagGems operator library and contribute code." +msgstr "参与 FlagGems 算子库共建并贡献代码" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:16 +msgid "" +"Prepare the dedicated FlagRelease machine and continuously publish model " +"images." +msgstr "FlagRelease 专机就绪,模型镜像持续发布" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:18 +msgid "Work with operating-system distributions and cloud partner communities." +msgstr "协作推进,带入操作系统发行版和云合作社区" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_intermediate.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_intermediate.mo new file mode 100644 index 0000000000..fd71c335ab Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_intermediate.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_intermediate.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_intermediate.po new file mode 100644 index 0000000000..0fb0f9344d --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/09_promotion_criteria/promotion_to_intermediate.po @@ -0,0 +1,50 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:38+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:2 +msgid "Requirements for promotion from basic to intermediate" +msgstr "初级到中级的必要条件" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:4 +msgid "The FlagTree chip branch runs stably for at least four weeks" +msgstr "FlagTree 芯片分支稳定运行 ≥4 周" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:6 +msgid "FlagGems operator test coverage is at least 80%" +msgstr "FlagGems 算子测试覆盖 ≥80%" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:8 +msgid "" +"The CI single machine is in the FlagOS resource pool with availability of" +" at least 99%" +msgstr "CI 单机已纳入 FlagOS 资源池,可用性 ≥99%" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:10 +# +msgid "The chip passes Sig repository functional validation." +msgstr "芯片通过 Sig 仓功能验证" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:12 +msgid "Establish an SBOM scanning mechanism" +msgstr "建立 SBOM 扫描机制" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:14 +msgid "The driver/SDK can be obtained unattended" +msgstr "驱动/SDK 可无人值守获取" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:16 +msgid "Sign and pass legal compliance review" +msgstr "签署并通过法务合规审查" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/10_risks_and_maintenance/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/10_risks_and_maintenance/index.mo new file mode 100644 index 0000000000..c605a651a4 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/10_risks_and_maintenance/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/10_risks_and_maintenance/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/10_risks_and_maintenance/index.po new file mode 100644 index 0000000000..c1717076e0 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud/10_risks_and_maintenance/index.po @@ -0,0 +1,101 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:56+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/10_risks_and_maintenance/index.md:2 +msgid "Risks and Maintenance" +msgstr "风险与注意事项" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "**Risk or requirement**" +msgstr "**风险项**" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "**Description**" +msgstr "**说明**" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "Driver access method" +msgstr "驱动获取方式" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "" +"Refreshing token authentication every 30 minutes is impractical for " +"automation. Driver and SDK packages should be openly downloadable or use " +"long-lived authorization." +msgstr "出于自动化考虑,每 30 分钟刷新一次 token 认证的方式不太现实,建议驱动/SDK 开放下载或使用长期有效授权。" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "Specific dependency trade-offs" +msgstr "特定依赖折衷" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "" +"Some trade-offs are allowed at the intermediate stage, but specific " +"dependencies must be eliminated before advanced cooperation." +msgstr "中级阶段允许保留部分折衷方案,但进入高级合作前必须消除特定依赖。" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "Image build neutrality" +msgstr "镜像构建中立性" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "" +"Use neutral base images to demonstrate compatibility rather than relying " +"entirely on vendor-customized images." +msgstr "建议使用中立基础镜像体现兼容性,而非完全依赖厂商定制镜像。" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "Long-term maintenance commitment" +msgstr "长期维护承诺" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "" +"Advanced cooperation requires continuous security maintenance and " +"progressive improvement of TLS and other version lifecycle management." +msgstr "高级合作要求对安全性问题承诺持续维护,并逐步完善 TLS 等版本生命周期管理。" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "Heterogeneous mixed efficiency" +msgstr "异构混合效率" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "" +"When a chip is used in a heterogeneous cluster at the intermediate stage," +" pay particular attention to mixed communication efficiency of at least " +"81%." +msgstr "中级阶段若芯片用于异构集群,需额外关注混合通信效率 ≥81%。" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "Device SLA" +msgstr "设备 SLA" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "" +"CI devices must stay online 24/7, SVT machines must be isolated, and the " +"FR machine must be configured independently. Long-term offline status may" +" cause certification downgrade or revocation." +msgstr "CI 设备要求 7×24 在线,SVT 隔离部署,FR 专机独立配置。长期离线可能导致认证降级或撤销。" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "Waiver period" +msgstr "豁免期限" + +#: ../../chip_adaptation_guide/cloud/10_risks_and_maintenance/index.md:1 +msgid "" +"Each certification-test waiver may last no more than six months. An " +"unresolved waiver must be resubmitted or may result in certification " +"downgrade." +msgstr "认证测试中的每项豁免最长 6 个月,到期未修复需重新申请或接受认证降级。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud_adaptation_guide_index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud_adaptation_guide_index.mo new file mode 100644 index 0000000000..460cbe1166 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud_adaptation_guide_index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud_adaptation_guide_index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud_adaptation_guide_index.po new file mode 100644 index 0000000000..34746d3352 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/cloud_adaptation_guide_index.po @@ -0,0 +1,41 @@ +# Translations template for PROJECT. +# Copyright (C) 2026 ORGANIZATION +# This file is distributed under the same license as the PROJECT project. +# FIRST AUTHOR , 2026. +msgid "" +msgstr "" +"Project-Id-Version: PROJECT VERSION\n" +"Report-Msgid-Bugs-To: EMAIL@ADDRESS\n" +"POT-Creation-Date: 2026-09-21 19:07+0800\n" +"PO-Revision-Date: YEAR-MO-DA HO:MI+ZONE\n" +"Last-Translator: FULL NAME \n" +"Language-Team: LANGUAGE \n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/cloud_adaptation_guide_index.md:1 +msgid "FlagOS Southbound Chip Adaptation Guide: Cloud Chips" +msgstr "FlagOS 南向芯片适配指南:云端芯片" + +#: ../../chip_adaptation_guide/cloud_adaptation_guide_index.md:3 +msgid "" +"This guide is for cloud AI chip vendors, FlagOS adaptation engineers, " +"framework developers, and distribution partners. It describes the " +"complete path for integrating cloud chips with FlagOS, including " +"cooperation levels, hardware resources, the base environment, compiler " +"and operator adaptation, communication and training/inference framework " +"adaptation, compatibility certification, distribution validation, quality" +" metrics, automation tools, promotion criteria, and risk management." +msgstr "" +"本手册面向云端 AI 芯片厂商、FlagOS 适配工程师、框架开发者和发行版合作伙伴,介绍云端芯片接入 FlagOS 的完整路径。" +"内容覆盖合作分级、硬件资源、基础环境、编译器与算子适配、通信与训练/推理框架适配、兼容性认证、发行版验证、质量指标、" +"自动化工具、晋升条件和风险管理。" + +#: ../../chip_adaptation_guide/cloud_adaptation_guide_index.md:5 +msgid "" +"Cloud adaptation focuses on multi-card and multi-node operation, cluster " +"communication, heterogeneous interoperability, and large-model training " +"and inference." +msgstr "云端适配重点关注多卡与多节点运行、集群通信、异构互操作以及大模型训练与推理。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/01_technical_stack/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/01_technical_stack/index.mo new file mode 100644 index 0000000000..11d0eb3ccd Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/01_technical_stack/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/01_technical_stack/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/01_technical_stack/index.po new file mode 100644 index 0000000000..5aaf5f8a85 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/01_technical_stack/index.po @@ -0,0 +1,36 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/01_technical_stack/index.md:2 +msgid "FlagOS Technology Stack Overview" +msgstr "FlagOS 技术堆栈全景" + +#: ../../chip_adaptation_guide/_shared/01_technical_stack/index.md:4 +msgid "The figure below shows the latest FlagOS technology stack. The FlagOS ecosystem currently includes three categories of open-source projects. The FlagOS open-source core libraries include the FlagGems general-purpose large-model and domain-specific operator libraries, the FlagScale training and inference framework, the FlagTree unified compiler, and the FlagCX unified communication library. We also define ecosystem-enabling projects, including plugins for training and inference engines and open-source tools designed by the FlagOS community, such as KernelGen for automatic operator generation, FlagRelease for automatic migration and release, and FlagPerf for multi-chip evaluation." +msgstr "下图展示 FlagOS 最新技术堆栈全景。FlagOS生态目前包括三大开源项目门类,分别为FlagOS开源核心库:包括了FlagGems通用大模型算子库,多领域算子库;FlagScale训练推理框架,FlagTree统一编译器和FlagCX统一通信库。同时我们定义了FlagOS生态使能项目,对接训练推理引擎的插件,以及FlagOS社区主导设计的开源工具,包括KernelGen算子自动生成,FlagRelease自动迁移发版和FlagPerf多芯评测工具。" + +#: ../../chip_adaptation_guide/_shared/01_technical_stack/index.md:6 +msgid "![image.png](../../../images/flagos-architecture-en.png)" +msgstr "![image.png](../../../images/flagos-architecture-zh.png)" + +#: ../../chip_adaptation_guide/_shared/01_technical_stack/index.md:6 +msgid "image.png" +msgstr "image.png" + +#: ../../chip_adaptation_guide/_shared/01_technical_stack/index.md:8 +msgid "Figure 1. FlagOS technology stack overview" +msgstr "图 1 FlagOS 技术堆栈全景" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/02_cooperation_model/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/02_cooperation_model/index.mo new file mode 100644 index 0000000000..1c3c82c8bf Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/02_cooperation_model/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/02_cooperation_model/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/02_cooperation_model/index.po new file mode 100644 index 0000000000..8d1f7d9ec4 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/02_cooperation_model/index.po @@ -0,0 +1,151 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-20 00:00+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/02_cooperation_model/edge.md:2 +msgid "Cooperation Level Model" +msgstr "合作分级模型" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Cooperation is divided into three levels: basic, intermediate, and advanced. Chip companies progress through the levels step by step. Given the characteristics of edge chips, no adaptation is required for the FlagCX communication library or the large-model training and inference framework." +msgstr "合作分为初级、中级、高级三个级别,芯片公司逐级晋升。根据端侧芯片的实际情况,对通信库FlagCX,大模型训推框架无适配要求。" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "**Dimension**" +msgstr "**维度**" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "**Basic (integration validation)**" +msgstr "**初级(接入验证)**" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "**Intermediate (deep adaptation)**" +msgstr "**中级(深度适配)**" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "**Advanced (ecosystem co-building)**" +msgstr "**高级(生态共建)**" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Positioning" +msgstr "定位" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Integration validation" +msgstr "接入验证" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Deep adaptation" +msgstr "深度适配" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Ecosystem co-building" +msgstr "生态共建" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Operators" +msgstr "算子" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Adopt selected high-performance FlagGems operators" +msgstr "采纳部分 FlagGems 高性能算子" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Fully support the FlagGems Stable operator set and commit to using FlagGems by default" +msgstr "芯片全面支持 FlagGems 稳定(Stable)算子集合,承诺默认采用 FlagGems" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Co-build FlagGems and generate specialized operators with KernelGen
" +msgstr "参与 FlagGems 共建,KernelGen 算子特化生成
" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Compiler" +msgstr "编译器" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Use the FlagTree compiler and C++ Wrapper" +msgstr "采用 FlagTree 编译器 + C++ Wrapper" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Optimize with Hints and TLE; use FlagTree as the default compiler" +msgstr "采用 Hints 机制优化 + TLE,默认使用FlagTree 作为编译器" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Co-build all three Triton TLE implementation layers" +msgstr "全面共建Triton TLE三层实现路径" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "CI/CD" +msgstr "CI/CD" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Use the FlagOS CI resource pool for unit tests
" +msgstr "支持使用 FlagOS CI 资源池执行单元测试
" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Co-build test cases, upstream code, and provide machine resources. Jointly solve technical issues and complete model adaptation and optimization" +msgstr "共建测试用例,代码进主线,提供机器资源。共同技术攻关,完成模型适配与优化" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Bring in operating-system and cloud partner communities" +msgstr "带入操作系统和云合作社区" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Compatibility and security" +msgstr "兼容性与安全" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Provide base container image materials and download channels for customized third-party dependencies" +msgstr "提供基础容器镜像物料;提供定制第三方依赖包下载渠道" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Obtain FlagOS certification, promptly update customized packages, and respond immediately to package adaptation issues" +msgstr "设备获得 FlagOS 认证,及时升级定制软件包;对软件包适配问题承诺第一时间响应" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Share security working-group results and bring them into downstream distributions
" +msgstr "共享安全工作组成果,下游发行版带入
" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Open-source community" +msgstr "开源社区" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Basic member" +msgstr "初级会员" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Intermediate member, eligible to run for PMC" +msgstr "中级会员,可竞选 PMC" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Advanced member, with course and case-study coverage" +msgstr "高级会员,课程/案例覆盖" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Business expansion" +msgstr "商业拓展" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "—" +msgstr "—" + +#: ../../chip_adaptation_guide/edge/02_cooperation_model/index.md:8 +msgid "Bring in partner clouds and MaaS platforms" +msgstr "带入合作云、MaaS 平台" + +msgid "Bring in upstream and downstream ecosystem communities" +msgstr "上下游社区生态引入" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/common_constraints.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/common_constraints.mo new file mode 100644 index 0000000000..a4be5174ca Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/common_constraints.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/common_constraints.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/common_constraints.po new file mode 100644 index 0000000000..03973c6058 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/common_constraints.po @@ -0,0 +1,36 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/common_constraints.md:2 +msgid "Key Constraints" +msgstr "关键约束说明" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/common_constraints.md:4 +msgid "CI devices must stay online 24/7 and be managed through the unified FlagOS CI/CD resource pool." +msgstr "CI 设备要求 7×24 小时在线,纳入 FlagOS CI/CD 资源池统一调度。" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/common_constraints.md:6 +msgid "SVT devices should be physically isolated from CI devices to prevent validation jobs and CI jobs from competing for resources." +msgstr "SVT 设备应与 CI 设备物理隔离,避免验证任务与 CI 任务争抢资源。" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/common_constraints.md:8 +msgid "The dedicated FlagRelease machine should be configured independently and should not be shared with other workloads." +msgstr "FlagRelease 专机建议独立配置,不与其他工作负载混用。" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/common_constraints.md:10 +msgid "All devices must support automated access. Driver and SDK packages should preferably be available through open downloads or long-lived authorization." +msgstr "所有设备需满足自动化获取要求,驱动/SDK 获取方式优先开放下载或长期有效授权。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/device_evolution.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/device_evolution.mo new file mode 100644 index 0000000000..e9b6907ca0 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/device_evolution.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/device_evolution.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/device_evolution.po new file mode 100644 index 0000000000..f753e83cef --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/device_evolution.po @@ -0,0 +1,36 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md:2 +msgid "Device Investment Evolution" +msgstr "设备投入演进图" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md:5 +msgid "![image.png](/chip_adaptation_guide/assets/cloud/device_evolution-en.png)" +msgstr "![image.png](/chip_adaptation_guide/assets/cloud/device_evolution.png)" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md:5 ../../chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md:9 +msgid "image.png" +msgstr "image.png" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md:9 +msgid "![image.png](/chip_adaptation_guide/assets/edge/device_evolution-en.png)" +msgstr "![image.png](/chip_adaptation_guide/assets/edge/device_evolution.png)" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/device_evolution.md:12 +msgid "Figure 2. Device investment evolution" +msgstr "图 2 设备投入演进图" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/index.mo new file mode 100644 index 0000000000..632af71c4f Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/index.po new file mode 100644 index 0000000000..2db55c88dc --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/03_hardware_resources/index.po @@ -0,0 +1,168 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/index.md:2 +msgid "Hardware Resource Investment Plan" +msgstr "硬件资源投入计划" + +#: ../../chip_adaptation_guide/_shared/03_hardware_resources/edge.md:2 +msgid "Throughout chip adaptation, the chip company must provide physical devices at each stage to support CI/CD, validation, and release. For the first two batches, two machines are recommended for each batch. A separate machine is required for the FlagRelease stage. Additional machines are usually required for Day 0, distribution, cloud-native, or showroom plans. Virtualization can supplement multi-OS validation, but critical validation still requires physical devices." +msgstr "芯片适配全过程中,芯片公司需按阶段投入物理设备以支撑 CI/CD、验证和发布。资源补充说明:第一批建议 2 台机器,第二批建议 2 台机器;进入 FlagRelease 阶段需要一台单独机器。进入 Day 0、发行版、云原生或展厅展示计划时,通常需要额外机器。针对多 OS 操作系统验证,虚拟化技术可以作为补充,但关键验证仍需物理设备。" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "**Stage ID**" +msgstr "**阶段标识**" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "**Stage**" +msgstr "**阶段**" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "**Required devices**" +msgstr "**所需设备**" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "**Purpose**" +msgstr "**用途**" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "**Configuration requirements**" +msgstr "**配置要求**" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "**Availability timing**" +msgstr "**投入时间点**" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Stage 1" +msgstr "阶段 1" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Environment readiness" +msgstr "环境就绪" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "1 chip development board or server with remote access for the FlagOS team" +msgstr "芯片开发板/服务器,供 FlagOS 远程访问" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Single-card operation" +msgstr "单卡可运行,Ubuntu OS,64GB+ 内存" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Ubuntu OS, 64 GB or more memory" +msgstr "单卡可运行,Ubuntu OS,64GB+ 内存" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Prepare immediately after the Stage 0 MOU is signed" +msgstr "阶段 0 签署 MOU 后立即准备" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Stage 2" +msgstr "阶段 2" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "FlagTree compiler integration" +msgstr "FlagTree 编译器接入" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "1 CI machine added to the CI resource pool and 1 debug machine" +msgstr "CI 1 台 → 纳入 CI 资源池 + 调试机 1 台" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Compiler build validation and CI pipeline triggers" +msgstr "编译器构建验证、CI 流水线触发" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Single card, 200 GB or more disk space, network access to the FlagOS CI cluster" +msgstr "单卡,200GB+ 磁盘,网络可达 FlagOS CI 集群" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Connect to CI when Stage 2 starts" +msgstr "阶段 2 启动时接入 CI" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Stage 3" +msgstr "阶段 3" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "FlagGems operator adaptation" +msgstr "FlagGems 算子适配" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "2 CI machines and 1 debug machine" +msgstr "CI 2 台 + 调试机 1 台" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Automated CI tests and operator development/debugging" +msgstr "CI 自动化测试 + 算子开发调试" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "CI machines must stay reliably online; the debug machine should preferably be available to the FlagOS community" +msgstr "CI 机稳定在线,调试机尽量提供给 FlagOS 社区" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Confirm the CI machine SLA when Stage 3 starts" +msgstr "阶段 3 启动时确认 CI 机 SLA" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Stage 4" +msgstr "阶段 4" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "End-to-end validation" +msgstr "全流程验证" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "2 CI machines and 2 SVT machines" +msgstr "CI 2 台 + SVT 2 台" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Continuous CI regression and system validation testing" +msgstr "CI 持续回归 + 系统验证测试" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "SVT machines must be isolated from CI machines to avoid interference" +msgstr "SVT 机需与 CI 机隔离,避免相互干扰" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Available when Stage 4 starts" +msgstr "阶段 5 启动时到位" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Stage 5" +msgstr "阶段 5" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "FlagRelease release" +msgstr "FlagRelease 发布" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "1 dedicated FR machine" +msgstr "1 台 FR 专机" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Model migration, image builds, and release validation" +msgstr "模型迁移、镜像构建、发布验证" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Single card or above, 200 GB or more disk space, high-speed network" +msgstr "单卡及以上,200GB+ 磁盘,高速网络" + +#: ../../chip_adaptation_guide/edge/03_hardware_resources/index.md:8 +msgid "Available when Stage 5 starts" +msgstr "阶段 5 启动时到位" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/index.mo new file mode 100644 index 0000000000..892dfb5305 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/index.po new file mode 100644 index 0000000000..b7d94993bc --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/index.po @@ -0,0 +1,27 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-20 00:00+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/index.md:5 +msgid "Adaptation Stages and Execution Plan" +msgstr "适配阶段与执行计划" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/index.md:9 +msgid "The cloud chip path progresses through six stages from kickoff to advanced adaptation." +msgstr "芯片从起步到完成高级适配,按六个阶段推进。" + +msgid "The edge chip path progresses through the stages from kickoff to advanced adaptation." +msgstr "芯片从起步到完成高级适配,按阶段推进。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/progress_overview.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/progress_overview.mo new file mode 100644 index 0000000000..f2f481faa0 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/progress_overview.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/progress_overview.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/progress_overview.po new file mode 100644 index 0000000000..e4d97812dc --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/progress_overview.po @@ -0,0 +1,64 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:2 +msgid "Overall Progress Overview" +msgstr "全阶段进度总览" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:5 +msgid "**Overall progress overview:** The figure below shows the complete path from contract kickoff, environment readiness, compiler integration, operator coverage, communication, training and inference, validation, compliance, release, and promotion. It is organized around key deliverables, cooperation-level evolution, and hardware investment." +msgstr "**全阶段进度总览说明:** 下图从关键成果、合作等级演进和硬件投入三个维度,展示芯片从签约启动、环境就绪、编译器接入、算子覆盖、通信与训练推理,到验证合规和发布推广的完整路径。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:7 +msgid "**Performance metric requirements:** Starting from Stage 3, operator performance, communication throughput, model accuracy, and end-to-end efficiency gradually become acceptance metrics. For advanced cooperation, performance metrics are a key baseline. Under equivalent compute, the chip must reach a defined percentage of the NVIDIA baseline; the specific percentage will be confirmed later through the KT2 milestone." +msgstr "**性能指标要求:** 从阶段 3 起,算子性能、通信吞吐、模型精度和端到端效率逐步进入指标范围;高级合作会以性能指标作为重要基准,需要再同等算力下达到 NVIDIA 基线 X%,具体比例后续按 KT2期口径确定。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:11 +msgid "**Overall progress overview:** The figure below shows the complete path from contract kickoff, environment readiness, compiler integration, operator coverage, validation, compliance, release, and promotion. It is organized around key deliverables, cooperation-level evolution, and hardware investment." +msgstr "**全阶段进度总览说明:** 下图从关键成果、合作等级演进和硬件投入三个维度,展示芯片从签约启动、环境就绪、编译器接入、算子覆盖,到验证合规和发布推广的完整路径。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:13 +msgid "**Performance metric requirements:** Starting from Stage 3, operator performance, model accuracy, and end-to-end efficiency gradually become acceptance metrics. For advanced cooperation, performance metrics are a key baseline." +msgstr "**性能指标要求:** 从阶段 3 起,算子性能、模型精度和端到端效率逐步进入指标范围;高级合作会以性能指标作为重要基准。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:16 +msgid "**Quality and continuous maintenance requirements:** Vendors must commit to maintaining the mainline adaptation code. FlagOS upgrades and vendor Driver/SDK upgrades must both be covered by CI/CD validation so that performance results remain sustainable, reproducible, and releasable." +msgstr "**质量与持续维护要求:** 厂商需承诺维护主线适配代码,FlagOS 升级与厂商 Driver/SDK 升级需共同纳入 CI/CD 验证,确保性能结果可持续、可复测、可发布。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:19 +msgid "![image.png](/chip_adaptation_guide/assets/cloud/progress_overview-en.png)" +msgstr "![image.png](/chip_adaptation_guide/assets/cloud/progress_overview.png)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:19 ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:23 +msgid "image.png" +msgstr "image.png" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:23 +msgid "![image.png](/chip_adaptation_guide/assets/edge/progress_overview-en.png)" +msgstr "![image.png](/chip_adaptation_guide/assets/edge/progress_overview.png)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:26 +msgid "Figure 3. Overall progress overview" +msgstr "图 3 全阶段进度总览" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:29 +msgid "Estimated total duration: about 5-7 months from signing to release. Total hardware investment increases with the cooperation target. The basic path requires about 1 development board, 1 CI machine, 2 SVT machines, and 1 dedicated FR machine. Additional machines should be added for Day 0, distribution, cloud-native, or showroom plans." +msgstr "预计总周期:约 5-7 个月(从签约到发布)。硬件总投入随合作目标递进:基础路径约 1 台开发板 + 1 台 CI + 2 台 SVT + 1 台 FR 专机;进入 Day 0、发行版、云原生或展厅展示计划时按场景追加。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/progress_overview.md:33 +msgid "Estimated total duration: about 3-5 months from signing to release. Total hardware investment increases with the cooperation target. The basic path requires about 1 development board, 1 CI machine, 2 SVT machines, and 1 dedicated FlagRelease (FR) machine. Additional machines should be added for Day 0, distribution, cloud-native, or showroom plans." +msgstr "预计总周期:约 3-5个月(从签约到发布)。硬件总投入随合作目标递进:基础路径约 1 台开发板 + 1 台 CI + 2 台 SVT + 1 台 FlagRelease(FR) 专机;进入 Day 0、发行版、云原生或展厅展示计划时按场景追加。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_00_business.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_00_business.mo new file mode 100644 index 0000000000..af53498222 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_00_business.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_00_business.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_00_business.po new file mode 100644 index 0000000000..66a1561009 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_00_business.po @@ -0,0 +1,105 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 15:50+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:2 +msgid "Stage 0: Business Kickoff and Contracting" +msgstr "阶段 0:商务启动与签约" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:4 +msgid "**Goal**" +msgstr "**目标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:6 +msgid "Establish the cooperation relationship and clarify the responsibilities of both parties." +msgstr "确立合作关系,明确双方权责。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:8 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:10 +msgid "The chip company has a base SDK/driver and can provide a development environment." +msgstr "芯片公司具备基础 SDK/驱动,可提供开发环境" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:12 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:14 +msgid "Sign an MOU defining the cooperation goals, adaptation scope, and division of responsibilities." +msgstr "签署 MOU,明确合作目标、适配范围、责任分工" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:15 +msgid "Connect with the FlagOS cooperation manager and assign contacts for both parties." +msgstr "对接 FlagOS 合作经理,分配双方接口人" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:16 +msgid "Set the initial target level for chip adaptation, usually starting at the intermediate level." +msgstr "确定芯片适配的初始目标级别(通常从中级起步)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:17 +msgid "Sign the Contributor License Agreement (CLA)." +msgstr "签署贡献者许可协议(CLA)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:19 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:21 +msgid "Signed MOU and contact list for both parties." +msgstr "已签署的 MOU 文档,双方接口人通讯录" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:23 +msgid "**FlagOS contacts**" +msgstr "**FlagOS 方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:25 +msgid "Dedicated cooperation manager and technical architect." +msgstr "专属合作经理、技术架构师" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:27 +msgid "**Chip company contacts**" +msgstr "**芯片方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:29 +msgid "Project director and technical lead." +msgstr "项目总监、技术负责人" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:31 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:33 +msgid "MOU signed and contacts confirmed by both parties." +msgstr "MOU 签署完成,双方接口人确认" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:35 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:37 +msgid "1-2 weeks" +msgstr "1-2 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:39 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_00_business.md:41 +msgid "MOU template, CLA, and cooperation handbook" +msgstr "MOU 模板、CLA 协议、合作手册" + diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_01_environment.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_01_environment.mo new file mode 100644 index 0000000000..aa6f06fd16 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_01_environment.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_01_environment.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_01_environment.po new file mode 100644 index 0000000000..efe6904827 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_01_environment.po @@ -0,0 +1,142 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:2 +msgid "" +"Stage 1: Development Board Readiness and Base Environment Adaptation " +"(Chip-Level)" +msgstr "阶段 1:开发板就绪与基础环境适配(芯片级)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:4 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:6 +msgid "The MOU has been signed." +msgstr "MOU 已签署" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:8 +msgid "**Hardware investment**" +msgstr "**硬件投入**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:10 +msgid "" +"1 development board or server, with remote access provided by the vendor " +"to the FlagOS team." +msgstr "1 台开发板/服务器,厂商提供远程访问权限给 FlagOS 团队" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:12 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:14 +msgid "" +"Provide the chip development board and select at least one operating " +"system, preferably Ubuntu." +msgstr "提供芯片开发板,选择不少于一个操作系统(以 Ubuntu 为首选)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:15 +msgid "" +"Provide the driver and SDK access method. Open downloads are preferred. " +"If authorization-only downloads are used, programmatic authentication " +"must support CI/CD automation." +msgstr "提供驱动及 SDK 获取方式(推荐开放下载;若仅支持授权下载,需支持 CI/CD 自动化场景下的程序化认证)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:16 +msgid "" +"Confirm base image compatibility. A neutral image is preferred, followed " +"by a vendor-provided image." +msgstr "确认基础镜像兼容性(优先中立镜像,其次厂商提供)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:17 +msgid "" +"Confirm software compatibility, including kernel, Python, and PyTorch " +"community version coverage." +msgstr "确认软件兼容性:内核、Python、PyTorch 社区版本范围覆盖" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:18 +msgid "" +"Validate the base environment: PyTorch can recognize the target device " +"backend, and the equivalent device discovery API returns an available " +"device." +msgstr "基础环境验证:PyTorch 可识别目标设备后端,等效设备发现接口返回可用" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:20 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:22 +msgid "Environment readiness report and driver/SDK access documentation." +msgstr "环境就绪确认报告,驱动/SDK 获取方式文档" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:24 +msgid "**FlagOS contacts**" +msgstr "**FlagOS 方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:26 +msgid "Platform engineering team." +msgstr "平台工程团队" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:28 +msgid "**Chip company contacts**" +msgstr "**芯片方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:30 +msgid "Driver and firmware engineers." +msgstr "驱动/固件工程师" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:32 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:34 +msgid "" +"The chip can be correctly identified by the OS, for example through lspci" +" or dmidecode." +msgstr "芯片可被 OS 正确识别(lspci / dmidecode 等)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:35 +msgid "PyTorch can recognize the chip backend." +msgstr "PyTorch 可识别芯片后端" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:36 +msgid "" +"Basic operators, such as matrix multiplication and elementwise " +"operations, execute correctly." +msgstr "基础算子(矩阵乘、元素级操作)可正常执行" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:37 +msgid "The driver can be obtained unattended." +msgstr "驱动可无人值守获取" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:39 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:41 +msgid "1-2 weeks" +msgstr "1-2 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:43 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_01_environment.md:45 +msgid "" +"Environment configuration guide, driver installation manual, and " +"compatibility matrix." +msgstr "环境配置指南、驱动安装手册、兼容性矩阵" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_02_flagtree.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_02_flagtree.mo new file mode 100644 index 0000000000..0cff19e226 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_02_flagtree.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_02_flagtree.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_02_flagtree.po new file mode 100644 index 0000000000..55eef106de --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_02_flagtree.po @@ -0,0 +1,157 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:2 +msgid "" +"Stage 2: Compiler Adaptation - FlagTree Branch Integration (Entry-Level " +"Threshold)" +msgstr "阶段 2:编译器适配 — FlagTree 分支接入(初级门槛)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:4 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:6 +msgid "" +"The development board environment is ready, and the chip Triton backend " +"or compatibility layer is available." +msgstr "开发板环境就绪,芯片 Triton 后端(或兼容层)可用" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:8 +msgid "**Hardware investment**" +msgstr "**硬件投入**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:10 +msgid "" +"Add the Stage 1 device to the FlagOS CI resource pool and keep it online " +"24/7." +msgstr "阶段 1 的 1 台设备纳入 FlagOS CI 资源池,要求 7×24 在线" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:12 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:14 +msgid "Create a protected chip branch in the FlagTree repository." +msgstr "在 FlagTree 仓库中创建芯片受保护分支" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:15 +msgid "" +"The chip company provides adaptation code or a compatibility layer for " +"the Triton backend." +msgstr "芯片公司提供 Triton 后端的适配代码或兼容层" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:16 +msgid "" +"Integrate a C++ Wrapper if a bridge from Triton IR to the vendor SDK is " +"required." +msgstr "集成 C++ Wrapper(若需要桥接 Triton IR 到厂商 SDK)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:17 +msgid "Complete instruction mapping through the unified FLIR IR layer." +msgstr "通过 FLIR 统一 IR 层完成指令映射" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:18 +msgid "Configure a basic CI pipeline for automatic builds and compilation tests." +msgstr "配置基础 CI 流水线:自动构建 + 编译测试" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:20 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:22 +msgid "Available FlagTree chip branch and running basic CI pipeline." +msgstr "FlagTree 芯片分支可用,基础 CI 流水线运行" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:24 +msgid "**FlagOS contacts**" +msgstr "**FlagOS 方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:26 +msgid "FlagTree compiler team." +msgstr "FlagTree 编译器团队" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:28 +msgid "**Chip company contacts**" +msgstr "**芯片方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:30 +msgid "Compiler and toolchain engineers." +msgstr "编译器/工具链工程师" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:32 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:34 +msgid "Triton kernels can be compiled successfully to the chip backend." +msgstr "Triton kernel 可成功编译到芯片后端" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:35 +msgid "At least 10 basic operators can be compiled and executed correctly." +msgstr "至少 10 个基础算子可正确编译和执行" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:36 +msgid "" +"CI build pass rate is at least 95%, and machine availability is at least " +"99%." +msgstr "CI 构建通过率 ≥95%,机器可用性 ≥99%" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:38 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:40 +msgid "2-4 weeks" +msgstr "2-4 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:42 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_02_flagtree.md:44 +msgid "" +"[FlagTree integration " +"guide](https://jwolpxeehx.feishu.cn/docx/VIridPa16odD9hxViDLcsD2vnXb): " +"integration plans for GPGPU and DSA/NPU devices.
Triton-TLE: [GitHub " +"documentation](https://github.com/flagos-ai/FlagTree/wiki/TLE), " +"[reference NV implementation](https://github.com/flagos-" +"ai/FlagTree/tree/triton_v3.6.x), [TLE vendor PR " +"template](https://jwolpxeehx.feishu.cn/file/VIOsbkSYZoI3I5xhxdLcuz8snVg?from=from_copylink)," +" [TLE raw CUDA integration vendor " +"example](https://jwolpxeehx.feishu.cn/wiki/Ipvbwa1WYintOwkmNzUchJJun0c).
C++" +" Wrapper: [libtriton_jit documentation](https://github.com/flagos-" +"ai/libtriton_jit), [sample PR](https://github.com/flagos-" +"ai/libtriton_jit/pull/12/changes), [C++ Runtime multi-backend and Triton " +"operator development " +"sharing](https://jwolpxeehx.feishu.cn/slides/RPkms0jYol2TA7dvbfScjVJWnlu?from=from_copylink)." +msgstr "" +"[FlagTree " +"接入指南](https://jwolpxeehx.feishu.cn/docx/VIridPa16odD9hxViDLcsD2vnXb): 包含 " +"GPGPU、DSA/NPU 两类设备的接入方案
Triton-TLE:[GitHub(说明文档)](https://github.com" +"/flagos-ai/FlagTree/wiki/TLE) [参考 NV 具体实现的代码](https://github.com/flagos-" +"ai/FlagTree/tree/triton_v3.6.x) [TLE 厂商 PR " +"模版](https://jwolpxeehx.feishu.cn/file/VIOsbkSYZoI3I5xhxdLcuz8snVg?from=from_copylink)" +" [TLE-Raw CUDA " +"接入-厂商示例说明](https://jwolpxeehx.feishu.cn/wiki/Ipvbwa1WYintOwkmNzUchJJun0c)
C++" +" Wrapper:[libtriton_jit 说明文档](https://github.com/flagos-ai/libtriton_jit)" +" [寒武纪示例](https://github.com/flagos-ai/libtriton_jit/pull/12/changes) [C++" +" Runtime 多后端与 Triton " +"算子开发分享](https://jwolpxeehx.feishu.cn/slides/RPkms0jYol2TA7dvbfScjVJWnlu?from=from_copylink)" + +#~ msgid "Stage 2: FlagTree Compiler Integration" +#~ msgstr "阶段 2:FlagTree 编译器接入" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_03_flaggems.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_03_flaggems.mo new file mode 100644 index 0000000000..e57c514207 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_03_flaggems.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_03_flaggems.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_03_flaggems.po new file mode 100644 index 0000000000..e3aa8712f9 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_03_flaggems.po @@ -0,0 +1,147 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:2 +msgid "" +"Stage 3: Operator Adaptation - FlagGems Coverage (Entry-to-Intermediate " +"Progression)" +msgstr "阶段 3:算子适配 — FlagGems 覆盖(初级→中级推进)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:4 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:6 +msgid "The FlagTree branch is available, and the chip Triton backend is stable." +msgstr "FlagTree 分支可用,芯片 Triton 后端稳定" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:8 +msgid "**Hardware investment**" +msgstr "**硬件投入**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:10 +msgid "1 CI machine reused from Stage 2 and 1 optional development/debug machine." +msgstr "CI 机 1 台(复用阶段 2)+ 开发调试机 1 台(可选)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:12 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:14 +msgid "" +"Confirm that the CI machine is connected to the FlagOS CI/CD resource " +"pool." +msgstr "确认 CI 机接入 FlagOS CI/CD 资源池" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:15 +msgid "Run the FlagGems test suite and identify failed or incompatible operators." +msgstr "运行 FlagGems 测试套件,识别失败/不兼容算子" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:16 +#, python-format +msgid "" +"Fix operator compatibility issues, targeting coverage of at least 80% of " +"core operators." +msgstr "修复算子兼容性问题,目标覆盖 ≥80% 核心算子" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:17 +msgid "" +"Use KernelGen to automatically generate missing operators and validate " +"them on the chip." +msgstr "利用 KernelGen 自动生成缺失算子,并针对该芯片验证" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:18 +#, python-format +msgid "" +"Tune operator performance, targeting at least 80% of the vendor native " +"operator performance." +msgstr "算子性能调优,目标达到或超过厂商原生算子性能的 80%" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:19 +msgid "Establish an SBOM scanning mechanism." +msgstr "建立 SBOM 扫描机制" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:21 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:23 +msgid "" +"FlagGems operator compatibility report, test coverage data, and SBOM " +"scanning baseline." +msgstr "FlagGems 算子兼容性报告,测试覆盖率数据,SBOM 扫描基线" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:25 +msgid "**FlagOS contacts**" +msgstr "**FlagOS 方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:27 +msgid "FlagGems operator team and KernelGen platform team." +msgstr "FlagGems 算子团队 + KernelGen 平台团队" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:29 +msgid "**Chip company contacts**" +msgstr "**芯片方接口人**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:31 +msgid "Operator development engineers." +msgstr "算子开发工程师" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:33 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:35 +msgid "FlagGems operator test pass rate is at least 80%." +msgstr "FlagGems 算子测试通过率 ≥80%" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:36 +msgid "Core LLM operators pass at 100%." +msgstr "核心 LLM 算子 100% 通过" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:37 +#, python-format +msgid "Operator performance reaches at least 80% of vendor native performance." +msgstr "算子性能 ≥ 厂商原生性能的 80%" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:38 +msgid "The CI test pipeline runs stably for at least two weeks." +msgstr "CI 测试流水线稳定运行 ≥2 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:40 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:42 +msgid "4-8 weeks" +msgstr "4-8 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:44 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_03_flaggems.md:46 +msgid "" +"[FlagGems operator library specification " +"(trial)](https://jwolpxeehx.feishu.cn/docx/GawHdXsuRomaQNxpISec30lxnFg)
[KernelGen" +" Wiki](https://docs.flagos.io/projects/kernelgen/en/latest/)" +msgstr "" +"[FlagGems算子库规范[试行]](https://jwolpxeehx.feishu.cn/docx/GawHdXsuRomaQNxpISec30lxnFg)
[KernelGen" +" Wiki](https://docs.flagos.io/projects/kernelgen/en/latest/)" + +#~ msgid "Stage 3: FlagGems Operator Adaptation" +#~ msgstr "阶段 3:FlagGems 算子适配" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_release.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_release.mo new file mode 100644 index 0000000000..c8569d924c Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_release.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_release.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_release.po new file mode 100644 index 0000000000..19b28125d2 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_release.po @@ -0,0 +1,146 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:2 +msgid "" +"Stage 5: FlagRelease Release and Commercial Promotion (Intermediate-to-" +"Advanced Progression)" +msgstr "阶段 5:FlagRelease 发布与商业推广(中级→高级)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:4 +msgid "" +"For edge chip adaptation, this page corresponds to Stage 5. Edge " +"adaptation has no cloud communication or training/inference framework " +"stage, so after Stage 4 End-to-End Validation and Legal Compliance " +"passes, it proceeds directly to FlagRelease release, image availability, " +"and commercial promotion." +msgstr "" +"端侧芯片适配中,本页对应阶段 5。端侧不设置云端的通信与训练推理框架阶段,因此在阶段 4 的全流程验证与法务合规通过后,直接进入 " +"FlagRelease 发布、镜像上线和商业推广。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:6 +msgid "**Stage ID**" +msgstr "**阶段标识**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:8 +msgid "Stage 5" +msgstr "阶段 5" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:10 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:12 +msgid "Mainline merge is complete, and certification tests have passed." +msgstr "主干合入完成,认证测试通过" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:14 +msgid "**Hardware investment**" +msgstr "**硬件投入**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:16 +msgid "" +"1 dedicated FlagRelease machine, configured independently and not shared " +"with CI or SVT." +msgstr "FlagRelease 专用机 1 台(独立配置,不与 CI/SVT 混用)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:18 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:20 +msgid "Obtain official FlagOS certification for the device." +msgstr "设备获得 FlagOS 正式认证" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:21 +msgid "Run model migration and image builds on the FlagRelease platform." +msgstr "FlagRelease 平台运行模型迁移和镜像构建" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:22 +msgid "Release Docker images for this chip, expanding the release list over time." +msgstr "发布该芯片的 Docker 镜像(按实际发布清单持续扩展)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:23 +msgid "Join the FlagOS large-model T+0 release plan." +msgstr "加入 FlagOS 大模型 T+0 发布计划" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:24 +msgid "Integrate with partner cloud and MaaS platforms." +msgstr "带入合作云平台和 MaaS 平台" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:25 +msgid "" +"Promote the vendor to intermediate or advanced open-source community " +"membership." +msgstr "上升为开源社区中级/高级会员" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:26 +msgid "Prepare joint branding exposure." +msgstr "联合品牌露出" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:28 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:30 +msgid "" +"FlagOS certification certificate, FlagRelease image list, and joint " +"branding plan." +msgstr "FlagOS 认证证书、FlagRelease 镜像清单、联合品牌方案" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:32 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:34 +msgid "FlagOS compatibility certification is passed." +msgstr "通过 FlagOS 兼容性认证" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:35 +msgid "" +"At least 10 chip-adapted model images are published on the FlagRelease " +"platform." +msgstr "FlagRelease 平台上线 ≥10 个模型的芯片适配镜像" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:36 +msgid "At least one joint customer case is delivered." +msgstr "至少 1 个联合客户案例落地" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:37 +msgid "Dedicated FlagRelease machine availability is at least 99%." +msgstr "FlagRelease 专用机可用性 ≥99%" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:39 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:41 +msgid "4-6 weeks" +msgstr "4-6 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:43 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_release_edge.md:45 +msgid "" +"Certification standard, image release specification, and joint branding " +"guide
[FlagRelease " +"Wiki](https://docs.flagos.io/projects/FlagRelease/en/latest/)" +msgstr "" +"认证标准文档、镜像发布规范、品牌联合指南
[FlagRelease " +"Wiki](https://docs.flagos.io/projects/FlagRelease/en/latest/)" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_validation.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_validation.mo new file mode 100644 index 0000000000..ffa23e6af1 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_validation.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_validation.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_validation.po new file mode 100644 index 0000000000..90ed465189 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_05_validation.po @@ -0,0 +1,139 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:2 +msgid "" +"Stage 4: End-to-End Validation and Legal Compliance (Intermediate Closing" +" Gate)" +msgstr "阶段 4:全流程验证与法务合规(中级收尾关卡)" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:4 +msgid "" +"For edge chip adaptation, this page corresponds to Stage 4. Edge " +"adaptation has no independent communication framework stage, so it enters" +" End-to-End Validation and Legal Compliance directly after Stage 3. This " +"page reuses the same validation topic but displays Stage 4 for the edge " +"process." +msgstr "" +"端侧芯片适配中,本页对应阶段 4。端侧不设置独立的通信框架阶段,因此阶段 3 " +"之后直接进入全流程验证与法务合规;本页复用同一验证主题,但阶段标识按端侧流程显示为阶段 4。" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:6 +msgid "**Stage ID**" +msgstr "**阶段标识**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:8 +msgid "Stage 4" +msgstr "阶段 4" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:10 +msgid "**Prerequisites**" +msgstr "**前置条件**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:12 +msgid "" +"Technical adaptation is complete for Stages 2-3, and the CI/CD resource " +"pool is ready." +msgstr "技术适配完成(阶段 2-3),CI/CD 资源池就绪" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:14 +msgid "**Hardware investment**" +msgstr "**硬件投入**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:16 +msgid "" +"2 CI machines and 2 SVT (system validation testing) machines. SVT " +"machines must be physically isolated from CI machines." +msgstr "CI 机 2 台 + SVT(系统验证测试)机 2 台;SVT 与 CI 机物理隔离" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:18 +msgid "**Activities**" +msgstr "**执行事项**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:20 +msgid "Validate SIG repository functionality." +msgstr "Sig 仓功能验证" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:21 +msgid "Run DevSecOps automated build, deployment, and validation." +msgstr "DevSecOps 自动化构建、部署、验证" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:22 +msgid "Validate SBOM scanning." +msgstr "SBOM 扫描验证" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:23 +msgid "Run FlagOS CTS compatibility certification tests." +msgstr "运行 FlagOS CTS 兼容性认证测试" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:24 +msgid "Perform legal compliance review." +msgstr "法务合规审查" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:26 +msgid "**Deliverables**" +msgstr "**交付物**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:28 +msgid "" +"Functional validation report, SBOM report, CTS test report, and legal " +"compliance review opinion." +msgstr "功能验证报告、SBOM 报告、CTS 测试报告、法务合规审查意见书" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:30 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:32 +msgid "SIG repository test cases pass at 100%." +msgstr "Sig 仓测试用例 100% 通过" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:33 +msgid "All CTS test items pass or have an explicit waiver list." +msgstr "CTS 测试项全部通过或有明确豁免清单" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:34 +msgid "SBOM scanning reports no critical vulnerabilities." +msgstr "SBOM 扫描无高危漏洞" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:35 +msgid "Legal review has no unresolved items." +msgstr "法务审查无未解决项" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:36 +msgid "" +"CI/CD and SVT end-to-end automation runs for at least seven days without " +"interruption." +msgstr "CI/CD + SVT 全流程自动化运行 ≥7 天无中断" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:38 +msgid "**Estimated duration**" +msgstr "**预计周期**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:40 +msgid "3-4 weeks" +msgstr "3-4 周" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:42 +msgid "**Related documents**" +msgstr "**对应文档**" + +#: ../../chip_adaptation_guide/_shared/04_adaptation_plan/stage_05_validation_edge.md:44 +msgid "" +"SIG repository test specification, CTS specification, SBOM standard, " +"compliance review checklist, and DevSecOps practice guide." +msgstr "Sig 仓测试规范、CTS 规范、SBOM 标准、合规审查清单、DevSecOps 实践指南" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.mo new file mode 100644 index 0000000000..2444c495b6 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.po new file mode 100644 index 0000000000..9230bbac8c --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.po @@ -0,0 +1,55 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:17+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.md:1 +msgid "Stage 6: Image Build" +msgstr "阶段 6:Image 构建" + +#: ../../chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.md:3 +msgid "" +"Image builds are divided into three levels: base image, runtime image, " +"and application image." +msgstr "镜像的构建分为三个层级:base image、runtime image、application image。" + +#: ../../chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.md:5 +msgid "Follow these documentation guides:" +msgstr "可遵循以下文档指引:" + +#: ../../chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.md:7 +msgid "" +"[Release Info Overview](https://flagos-ai.github.io/release-" +"info/overview/)" +msgstr "[Release Info 概览](https://flagos-ai.github.io/release-info/overview/)" + +#: ../../chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.md:8 +msgid "" +"[Release Info Onboarding Guide](https://flagos-ai.github.io/release-" +"info/contribution/onboarding/)" +msgstr "" +"[Release Info Onboarding 指南](https://flagos-ai.github.io/release-" +"info/contribution/onboarding/)" + +#: ../../chip_adaptation_guide/edge/04_adaptation_plan/stage_06_image_build.md:10 +msgid "" +"If you would like to join, submit an issue in the [release-info " +"repository](https://github.com/flagos-ai/release-info/issues)." +msgstr "" +"如有意向加入,可在 [release-info 仓库提交 issue](https://github.com/flagos-ai/release-" +"info/issues)。" + +#~ msgid "Image Build (Stage 6)" +#~ msgstr "Image 构建(阶段 6)" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_architecture.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_architecture.mo new file mode 100644 index 0000000000..d6d0f72eba Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_architecture.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_architecture.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_architecture.po new file mode 100644 index 0000000000..1411d2b5f4 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_architecture.po @@ -0,0 +1,32 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_architecture.md:2 +msgid "Certification Architecture" +msgstr "认证架构" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_architecture.md:4 +msgid "![image5.png](/chip_adaptation_guide/assets/common/certification_architecture-en.png)" +msgstr "![image5.png](/chip_adaptation_guide/assets/common/certification_architecture.png)" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_architecture.md:4 +msgid "image5.png" +msgstr "image5.png" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_architecture.md:6 +msgid "Figure 4. FlagOS certification architecture" +msgstr "图 4 FlagOS 认证架构" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.mo new file mode 100644 index 0000000000..ab4c29939a Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.po new file mode 100644 index 0000000000..6227c9fbfd --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.po @@ -0,0 +1,101 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:38+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_levels.md:2 +msgid "Certification Levels" +msgstr "四级认证等级" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "**Certification level**" +msgstr "**认证等级**" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "**Requirements**" +msgstr "**要求**" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "**Applicable objects**" +msgstr "**适用对象**" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "**Grant conditions**" +msgstr "**授予条件**" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "FlagOS Ready" +msgstr "FlagOS Ready" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "Basic compatibility certification" +msgstr "基础兼容认证" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "All chips entering intermediate cooperation" +msgstr "进入中级的所有芯片" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "Core CTS test items pass" +msgstr "CTS 核心测试项通过" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "FlagOS Verified" +msgstr "FlagOS Verified" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "Functionality, performance, and security" +msgstr "功能 + 性能 + 安全" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +# +msgid "Chips that complete intermediate adaptation" +msgstr "完成中级适配的芯片" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "CTS, VTS, and STS all pass" +msgstr "CTS + VTS + STS 全部通过" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "FlagOS Interoperable" +msgstr "FlagOS Interoperable" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "Heterogeneous interoperability certification" +msgstr "异构互操作认证" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "Chips entering advanced cooperation" +msgstr "进入高级的芯片" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "CTS, VTS, STS, and ITS all pass" +msgstr "CTS + VTS + STS + ITS 全部通过" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "FlagOS Platinum" +msgstr "FlagOS Platinum" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +# +msgid "Flagship certification" +msgstr "旗舰级认证" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "T+0 release partners" +msgstr "T+0 发布合作伙伴" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_levels.md:1 +msgid "All tests pass and the partner contributes to the community" +msgstr "全部测试通过 + 社区共建贡献" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_maintenance.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_maintenance.mo new file mode 100644 index 0000000000..9dc6d8309f Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_maintenance.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_maintenance.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_maintenance.po new file mode 100644 index 0000000000..0672b02955 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_maintenance.po @@ -0,0 +1,36 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-20 00:00+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_maintenance.md:4 +msgid "Certification Maintenance" +msgstr "认证维护" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_maintenance.md:6 +msgid "Validity period: 12 months; recertification is required at expiration." +msgstr "有效期:12 个月,到期重新认证。" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_maintenance.md:8 +msgid "Changes triggering retesting: major driver upgrades, Triton backend refactoring, or major PyTorch upgrades." +msgstr "变更触发重测:驱动大版本升级、Triton 后端重构、PyTorch 大版本升级。" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_maintenance.md:10 +#, python-format +msgid "Certification revocation: a CI pass rate below 80% for three consecutive months, or an unresolved critical security vulnerability." +msgstr "认证撤销机制:连续 3 个月 CI 通过率 <80%,或出现未修复的高危安全漏洞。" + +msgid "Waiver management: each waiver must state the technical reason, remediation plan, and expected remediation date; the maximum waiver period is six months." +msgstr "豁免管理:每项豁免需明确技术原因、修复计划、预计修复日期,最长豁免期 6 个月。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.mo new file mode 100644 index 0000000000..e3665e07c9 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.po new file mode 100644 index 0000000000..1302b7a066 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.po @@ -0,0 +1,90 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:07+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_objects.md:2 +msgid "Three-Tier Certification Objects" +msgstr "三层认证对象" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "**Certification tier**" +msgstr "**认证层级**" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "**Certification object**" +msgstr "**认证对象**" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "**Scope**" +msgstr "**覆盖范围**" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "**Benefits and listing**" +msgstr "**权益/展示**" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "Chip certification" +msgstr "芯片认证" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "Chip and base software stack" +msgstr "芯片和基础软件栈" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "" +"Drivers, SDK, compiler, operators, communication, frameworks, " +"performance, and security" +msgstr "驱动、SDK、编译器、算子、通信、框架、性能和安全" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "Chip compatibility certification, adaptation list, and Day 0 candidate" +msgstr "芯片兼容性认证、适配清单、Day 0 候选" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "System certification" +msgstr "整机认证" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "Complete server or device" +msgstr "服务器/设备整机" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "" +"Multi-card interconnect, thermal management, long-duration stability, " +"SVT, resource monitoring, and system stability" +msgstr "多卡互联、散热、长稳、SVT、资源监控、稳定性" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "System certification and showroom candidate" +msgstr "整机认证、展厅展示候选" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "Software/cloud-native certification" +msgstr "软件/云原生认证" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "OS distributions, containers, Kubernetes, and cloud-native platforms" +msgstr "OS 发行版、容器、Kubernetes、云原生平台" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "" +"Images, scheduling, Kubernetes, system integration, security, and " +"interoperability" +msgstr "镜像、调度、K8s、系统集成、安全和互操作" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/certification_objects.md:1 +msgid "Distribution plan and cloud-native ecosystem certification" +msgstr "发行版计划、云原生生态认证" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_process.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_process.mo new file mode 100644 index 0000000000..3846ab2a76 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_process.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_process.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_process.po new file mode 100644 index 0000000000..a05b8d56ae --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/certification_process.po @@ -0,0 +1,32 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_process.md:2 +msgid "Certification Process" +msgstr "认证流程" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_process.md:4 +msgid "![image6.png](/chip_adaptation_guide/assets/common/certification_process-en.jpeg)" +msgstr "![image6.png](/chip_adaptation_guide/assets/common/certification_process.png)" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_process.md:4 +msgid "image6.png" +msgstr "image6.png" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/certification_process.md:6 +msgid "Figure 5. Certification process" +msgstr "图 5 认证流程" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/index.mo new file mode 100644 index 0000000000..41659aad39 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/index.po new file mode 100644 index 0000000000..5f46e29b66 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/index.po @@ -0,0 +1,23 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-20 00:00+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/index.md:4 +msgid "FlagOS Compatibility Certification System" +msgstr "FlagOS 兼容性认证体系" + +msgid "Following the Android CDD/CTS design, FlagOS has established a tiered certification system. Based on further discussion, certification objects are divided into three layers: chips, complete systems, and software/cloud-native platforms, covering different vendor forms and later showroom, policy, and procurement scenarios." +msgstr "参考 Android CDD/CTS 设计,FlagOS 建立分级认证体系;同时结合会议讨论,认证对象进一步分为芯片、整机、软件/云原生三层,以覆盖不同厂商形态和后续展示、政策、集采场景。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/test_suites.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/test_suites.mo new file mode 100644 index 0000000000..77ee144c4a Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/test_suites.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/test_suites.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/test_suites.po new file mode 100644 index 0000000000..e138f8d50d --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/05_compatibility_certification/test_suites.po @@ -0,0 +1,263 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:07+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_edge.md:2 +msgid "Test Suites" +msgstr "测试套件详情" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_edge.md:4 +msgid "CTS (Compatibility Test Suite) - Compatibility Test Suite" +msgstr "CTS(Compatibility Test Suite)— 兼容性测试套件" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "**Test domain**" +msgstr "**测试域**" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "**Coverage**" +msgstr "**覆盖内容**" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "**Pass standard**" +msgstr "**通过标准**" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "Compiler compatibility" +msgstr "编译器兼容性" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "FlagTree compilation and FLIR IR mapping correctness" +msgstr "FlagTree 编译 + FLIR IR 映射正确性" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "100%" +msgstr "100%" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "Operator compatibility" +msgstr "算子兼容性" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "FlagGems core operator correctness" +msgstr "FlagGems 核心算子功能正确性" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid ">=95%" +msgstr "≥95%" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "Communication compatibility (not required for edge chips)" +msgstr "通信兼容性(端侧芯片不需适配)" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "FlagCX collective communication correctness" +msgstr "FlagCX 集合通信功能正确性" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "Framework compatibility (not required for edge chips)" +msgstr "框架兼容性(端侧芯片不需适配)" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "Basic FlagScale training and inference workflow" +msgstr "FlagScale 训练/推理基本流程" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "PyTorch API compatibility" +msgstr "PyTorch API 兼容性" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "Correctness of common aten API return values" +msgstr "常用 aten API 返回值正确性" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid ">=90%" +msgstr "≥90%" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "Triton API compatibility" +msgstr "Triton API 兼容性" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "Triton language feature support" +msgstr "Triton 语言特性支持度" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "Accuracy consistency" +msgstr "精度一致性" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "Comparison with the NVIDIA baseline" +msgstr "与 NVIDIA 基线对比,关键指标" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:5 +msgid "Deviation <2%" +msgstr "偏差 <2%" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_common.md:2 +msgid "VTS (Verification Test Suite) - Verification Test Suite" +msgstr "VTS(Verification Test Suite)— 验证测试套件" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Stress test" +msgstr "压力测试" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Continuous operator execution for 72 hours" +msgstr "连续 72 小时算子执行" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "No crashes" +msgstr "无崩溃" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Stability test" +msgstr "稳定性测试" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "1,000 repeated training and inference runs" +msgstr "重复 1000 次训练/推理流程" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Consistent results" +msgstr "结果一致" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Long-duration stability" +msgstr "长稳测试" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Continuous 24/7 operation" +msgstr "7×24 持续运行" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "No memory leaks" +msgstr "无内存泄漏" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Resource monitoring" +msgstr "资源监控" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "GPU utilization, memory bandwidth, and temperature" +msgstr "GPU 利用率、内存带宽、温度" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Within vendor specifications" +msgstr "在厂商规格范围内" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_edge.md:19 +msgid "STS (Security Test Suite) - Security Test Suite" +msgstr "STS(Security Test Suite)— 安全测试套件" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "SBOM scanning" +msgstr "SBOM 扫描" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Dependency vulnerability scanning" +msgstr "依赖组件漏洞扫描" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "No high-risk vulnerabilities or CVEs" +msgstr "无高危/CVE" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Driver security" +msgstr "驱动安全" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Driver file permission and signature validation" +msgstr "驱动文件权限、签名验证" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Passed" +msgstr "通过" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Container security" +msgstr "容器安全" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Container image scanning" +msgstr "镜像扫描" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "No high-risk vulnerabilities" +msgstr "无高危" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Communication encryption (not required for edge chips)" +msgstr "通信加密(端侧芯片不需适配)" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "FlagCX communication-link encryption validation" +msgstr "FlagCX 通信链路加密验证" + +#: ../../chip_adaptation_guide/_shared/05_compatibility_certification/test_suites_edge.md:28 +msgid "ITS (Interoperability Test Suite) - Interoperability Test Suite" +msgstr "ITS(Interoperability Test Suite)— 互操作性测试套件" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Heterogeneous mixed training" +msgstr "异构混合训练" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Chip and NVIDIA mixed training" +msgstr "该芯片 + NVIDIA 混合训练" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Efficiency >=81%" +msgstr "效率 ≥81%" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Heterogeneous mixed inference" +msgstr "异构混合推理" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Chip and NVIDIA mixed inference" +msgstr "该芯片 + NVIDIA 混合推理" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Accuracy deviation <2%" +msgstr "精度偏差 <2%" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Cross-vendor communication (not required for edge chips)" +msgstr "跨厂商通信(端侧芯片不需适配)" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "FlagCX cross-vendor testing" +msgstr "FlagCX 跨厂商测试" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Correctness 100%" +msgstr "正确性 100%" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "Multi-framework compatibility" +msgstr "多框架兼容" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "FlagScale, vLLM, and SGLang" +msgstr "FlagScale + vLLM + SGLang" + +#: ../../chip_adaptation_guide/edge/05_compatibility_certification/test_suites.md:16 +msgid "All passed" +msgstr "全部通过" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/ai_distribution.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/ai_distribution.mo new file mode 100644 index 0000000000..51f8f50518 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/ai_distribution.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/ai_distribution.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/ai_distribution.po new file mode 100644 index 0000000000..9db8c63f00 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/ai_distribution.po @@ -0,0 +1,84 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 15:50+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:2 +msgid "AI Distribution Validation" +msgstr "AI 发行版验证" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:4 +msgid "**Validation party**" +msgstr "**验证方**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:6 +msgid "AI distribution company" +msgstr "AI 发行版公司" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:8 +msgid "**Validation scope**" +msgstr "**验证范围**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:10 +msgid "Complete installation and operation of PyTorch on the chip." +msgstr "PyTorch 在该芯片上的完整安装和运行" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:11 +msgid "Loading and execution of the FlagGems operator library." +msgstr "FlagGems 算子库加载和执行" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:12 +msgid "End-to-end operation of the Triton compilation pipeline." +msgstr "Triton 编译流水线端到端工作" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:13 +msgid "Loading and inference of representative models such as Qwen2.5 and Llama3." +msgstr "代表模型(如 Qwen2.5、Llama3)加载和推理" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:15 +msgid "**Hardware requirements**" +msgstr "**硬件要求**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:17 +msgid "The dedicated machine provided by the chip company during the FlagRelease stage can be made available for remote use by the AI distribution." +msgstr "芯片公司在 FlagRelease 阶段提供的专用机可开放给 AI 发行版远程使用" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:19 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:21 +msgid "PyTorch core test cases pass." +msgstr "PyTorch 核心测试用例通过" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:22 +msgid "The FlagGems operator loading rate reaches 100%." +msgstr "FlagGems 算子加载率 100%" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:23 +msgid "Triton kernels compile and execute correctly." +msgstr "Triton kernel 编译和执行正确" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:25 +msgid "**Relationship to certification**" +msgstr "**与认证的关系**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/ai_distribution.md:27 +msgid "AI distribution validation is an important reference condition for FlagOS Platinum certification." +msgstr "AI 发行版验证是 FlagOS Platinum 认证的重要参考条件" + + + + + diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.mo new file mode 100644 index 0000000000..1c65b1a41d Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.po new file mode 100644 index 0000000000..45cc12a978 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.po @@ -0,0 +1,80 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:07+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/compatibility_declaration.md:2 +msgid "Compatibility Declaration" +msgstr "发行版兼容性声明" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "**Declaration level**" +msgstr "**声明等级**" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "**Meaning**" +msgstr "**含义**" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "**Condition**" +msgstr "**条件**" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "Platinum Partner" +msgstr "Platinum Partner" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "The chip is fully supported in the distribution." +msgstr "该芯片在发行版中完全受支持" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "Passes Level 1 + Level 2 + Level 3 validation and is an advanced partner." +msgstr "通过层级 1 + 2 + 3 验证,且为高级合作伙伴" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "Certified" +msgstr "Certified" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "The chip is certified to run normally in the distribution." +msgstr "该芯片在发行版中经认证可正常运行" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "Passes Level 1 + Level 2 validation." +msgstr "通过层级 1 + 2 验证" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "Compatible" +msgstr "Compatible" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "The chip is verified as compatible in the distribution." +msgstr "该芯片在发行版中经验证兼容" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "Passes at least Level 1 validation." +msgstr "至少通过层级 1 验证" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "Community Support" +msgstr "Community Support" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "Community-driven support without official distribution validation." +msgstr "社区驱动,发行版未做官方验证" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/compatibility_declaration.md:1 +msgid "Only FlagOS CTS has passed." +msgstr "仅通过 FlagOS CTS" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/index.mo new file mode 100644 index 0000000000..cb44129776 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/index.po new file mode 100644 index 0000000000..0044bff16e --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/index.po @@ -0,0 +1,23 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-20 00:00+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/index.md:4 +msgid "Distribution Validation System (Draft)" +msgstr "发行版公司验证体系(草稿)" + +msgid "End users obtain FlagOS capabilities through operating-system and AI distributions. Distribution companies must validate chip compatibility in their distributions so that adaptation results move beyond experimental environments into real delivery pipelines." +msgstr "芯片适配的最终用户通过操作系统发行版和 AI 发行版获取 FlagOS 能力。发行版公司需要在各自发行版中验证芯片兼容性,确保适配成果不仅停留在实验环境,而是进入真实交付链路。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/joint_validation.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/joint_validation.mo new file mode 100644 index 0000000000..b122767066 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/joint_validation.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/joint_validation.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/joint_validation.po new file mode 100644 index 0000000000..b7f104a7f5 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/joint_validation.po @@ -0,0 +1,106 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:07+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/joint_validation.md:2 +msgid "Joint Validation Process" +msgstr "发行版联合验证流程" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "**Step**" +msgstr "**步骤**" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "**Action**" +msgstr "**动作**" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "**Result**" +msgstr "**结果**" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "1" +msgstr "1" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "The chip passes FlagOS CTS" +msgstr "芯片通过 FlagOS CTS" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "CTS report and certification level" +msgstr "形成 CTS 报告和认证级别" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "2" +msgstr "2" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "" +"FlagOS sends the chip certification information to the distribution " +"company" +msgstr "FlagOS 向发行版公司推送芯片认证信息" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "Certification level, CTS report, and device access method" +msgstr "包含认证级别、CTS 报告、设备接入方式" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "3" +msgstr "3" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "OS distribution validation" +msgstr "OS 发行版验证" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "OS distribution validation report" +msgstr "生成 OS 发行版验证报告" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "4" +msgstr "4" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "AI distribution validation" +msgstr "AI 发行版验证" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "AI distribution validation report" +msgstr "生成 AI 发行版验证报告" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "5" +msgstr "5" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "ISV validation (optional)" +msgstr "ISV 验证(可选)" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "Third-party compatibility declaration" +msgstr "生成第三方兼容性声明" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "6" +msgstr "6" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "The distribution lists the chip" +msgstr "发行版收录芯片" + +#: ../../chip_adaptation_guide/edge/06_distribution_validation/joint_validation.md:1 +msgid "Entry in the \"FlagOS Compatible\" list" +msgstr "进入 “FlagOS Compatible” 列表成员" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/os_distribution.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/os_distribution.mo new file mode 100644 index 0000000000..75e38eb8ff Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/os_distribution.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/os_distribution.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/os_distribution.po new file mode 100644 index 0000000000..fedf2748cc --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/os_distribution.po @@ -0,0 +1,84 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 15:50+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:2 +msgid "OS Distribution Validation" +msgstr "OS 发行版验证" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:4 +msgid "**Validation party**" +msgstr "**验证方**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:6 +msgid "OS distribution company" +msgstr "操作系统发行版公司" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:8 +msgid "**Validation scope**" +msgstr "**验证范围**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:10 +msgid "Driver loading and stability on the target OS kernel version." +msgstr "芯片驱动在目标 OS 内核版本上的加载和稳定性" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:11 +msgid "Compatibility of core system libraries such as glibc, libstdc++, and OpenMPI." +msgstr "基础系统库(glibc、libstdc++、OpenMPI 等)兼容性" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:12 +msgid "Docker/containerd runtime operation." +msgstr "Docker/containerd 容器运行时正常工作" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:13 +msgid "Kernel module signature validation." +msgstr "内核模块签名验证" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:15 +msgid "**Hardware requirements**" +msgstr "**硬件要求**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:17 +msgid "One device provided by the chip company or made available for remote access." +msgstr "芯片公司提供或开放远程访问的 1 台设备" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:19 +msgid "**Acceptance criteria**" +msgstr "**检验指标**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:21 +msgid "The driver loads normally on the target kernel, with no errors in dmesg." +msgstr "驱动在目标内核上加载无异常(dmesg 无 error)" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:22 +msgid "Core LTP test items pass." +msgstr "LTP 核心测试项通过" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:23 +msgid "The container runtime can pull and run standard images." +msgstr "容器运行时可拉取并运行标准镜像" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:25 +msgid "**Relationship to certification**" +msgstr "**与认证的关系**" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/os_distribution.md:27 +msgid "After the chip obtains FlagOS Verified certification, it enters the OS distribution validation pipeline." +msgstr "芯片获得 FlagOS Verified 认证后,进入 OS 发行版验证管道" + + + + + diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/validation_levels.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/validation_levels.mo new file mode 100644 index 0000000000..094cfac4e5 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/validation_levels.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/validation_levels.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/validation_levels.po new file mode 100644 index 0000000000..c18e13ba9a --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/06_distribution_validation/validation_levels.po @@ -0,0 +1,32 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 15:57+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/validation_levels.md:2 +msgid "Distribution Validation Levels" +msgstr "发行版验证层级" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/validation_levels.md:4 +msgid "![image7.png](/chip_adaptation_guide/assets/common/distribution_validation-en.png)" +msgstr "![image7.png](/chip_adaptation_guide/assets/common/distribution_validation.png)" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/validation_levels.md:4 +msgid "image7.png" +msgstr "image7.png" + +#: ../../chip_adaptation_guide/_shared/06_distribution_validation/validation_levels.md:6 +msgid "Figure 6. Distribution validation levels" +msgstr "图 6 发行版验证层级" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/07_quality_metrics/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/07_quality_metrics/index.mo new file mode 100644 index 0000000000..dba18b9f47 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/07_quality_metrics/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/07_quality_metrics/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/07_quality_metrics/index.po new file mode 100644 index 0000000000..79ec2a153b --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/07_quality_metrics/index.po @@ -0,0 +1,155 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:38+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/07_quality_metrics/index.md:2 +msgid "Key Quality Metrics" +msgstr "关键检验指标" + +#: ../../chip_adaptation_guide/_shared/07_quality_metrics/edge.md:2 +# +msgid "" +"**Performance-related metrics must remain key acceptance criteria " +"throughout Stages 3-6.** Stage 3 focuses on operator performance, and " +"Stage 4 focuses on inference accuracy and end-to-end performance." +msgstr "" +"**性能相关指标需作为关键检验项贯穿阶段 3-5:** 阶段 3 关注算子性能,阶段 4 关注推理精度与端到端表现;阶段 5 关注 " +"FlagRelease 发布与镜像上线结果。" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "**Stage**" +msgstr "**阶段**" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "**Acceptance criteria**" +msgstr "**核心检验指标**" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "**Pass standard**" +msgstr "**通过标准**" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "**Required hardware**" +msgstr "**涉及硬件**" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "Stage 0" +msgstr "阶段 0" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "MOU signing" +msgstr "MOU 签署" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "Effective after both parties sign or seal" +msgstr "双方盖章生效" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "N/A" +msgstr "—" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "Stage 1" +msgstr "阶段 1" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "Environment readiness" +msgstr "环境就绪" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "Chip is recognized by the OS and PyTorch can recognize the backend" +msgstr "芯片被 OS + PyTorch 识别" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "1 development board" +msgstr "开发板 1 台" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "Stage 2" +msgstr "阶段 2" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "Triton kernel compilation pass rate" +msgstr "Triton kernel 编译通过率" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid ">=95%, CI machine availability >=99%" +msgstr "≥95%,CI 机器可用性 ≥99%" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "1 CI machine" +msgstr "CI 1 台" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "Stage 3" +msgstr "阶段 3" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "FlagGems operator test pass rate" +msgstr "FlagGems 算子测试通过率" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid ">=80%, core operators 100%" +msgstr "≥80%,核心算子 100%" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "2 CI machines and 1 debug machine" +msgstr "CI 2 台 + 调试机 1 台" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "Stage 4" +msgstr "阶段 4" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "Model accuracy deviation" +msgstr "模型精度偏差" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +#, python-format +msgid "<2% relative to the NVIDIA baseline" +msgstr "<2%(相对 NVIDIA 基线)" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "Stage 5" +msgstr "阶段 5" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "CTS core tests passed" +msgstr "CTS 核心测试项通过" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "100% passed or waiver approved" +msgstr "100% 或豁免获批" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "2 CI machines and 2 SVT machines" +msgstr "CI 2 台 + SVT 2 台" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +# +msgid "Stage 6" +msgstr "阶段 6" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "Number of models released through FlagRelease" +msgstr "FlagRelease 上线模型数" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid ">=10 models" +msgstr "≥10 个" + +#: ../../chip_adaptation_guide/edge/07_quality_metrics/index.md:8 +msgid "1 dedicated FR machine" +msgstr "FR 专机 1 台" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.mo new file mode 100644 index 0000000000..51130065b4 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.po new file mode 100644 index 0000000000..e478084343 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.po @@ -0,0 +1,125 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:07+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/08_tools_and_skills/skill_catalog.md:2 +msgid "Supporting Skills and Automation Tools" +msgstr "配套 Skill / 自动化工具支持" + +#: ../../chip_adaptation_guide/_shared/08_tools_and_skills/skill_catalog.md:4 +msgid "" +"FlagOS packages multiple adaptation workflow steps as reusable Agent " +"Skills. Chip companies and their development teams can use them according" +" to the adaptation stage." +msgstr "FlagOS 将适配流程的多个步骤封装为可复用的 Agent Skill,芯片公司及其开发团队可按阶段使用。" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "**Skill name**" +msgstr "**Skill 名称**" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "**Applicable stage**" +msgstr "**适用阶段**" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "**Function**" +msgstr "**功能**" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "gpu-container-setup-flagos" +msgstr "gpu-container-setup-flagos" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "Stage 1" +msgstr "阶段 1" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "" +"Automatically detect the GPU vendor, launch a PyTorch container, and " +"validate the GPU/accelerator card." +msgstr "自动检测 GPU 厂商,拉起 PyTorch 容器,验证 GPU/加速卡" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "install-stack-flagos" +msgstr "install-stack-flagos" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "Stages 1-2" +msgstr "阶段 1-2" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "Install the full stack: FlagTree, FlagGems, FlagCX, and FlagScale." +msgstr "全栈安装(FlagTree + FlagGems + FlagCX + FlagScale)" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "kernelgen-flagos" +msgstr "kernelgen-flagos" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "Stage 3" +msgstr "阶段 3" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "Automatically generate and iteratively optimize operators." +msgstr "算子自动生成与迭代优化" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "model-migrate-flagos" +msgstr "model-migrate-flagos" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "Stage 4b" +msgstr "阶段 4b" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "Backport models from upstream vLLM to vllm-plugin-FL." +msgstr "从上游 vLLM 回移植模型到 vllm-plugin-FL" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "model-verify-flagos" +msgstr "model-verify-flagos" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "Stage 5" +msgstr "阶段 5" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "Verify the serving stack step by step." +msgstr "分步验证 serving stack" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "perf-test-flagos" +msgstr "perf-test-flagos" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "Stages 4-5" +msgstr "阶段 4-5" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "Run performance benchmarks." +msgstr "性能基准测试" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "flagrelease-entrance-flagos" +msgstr "flagrelease-entrance-flagos" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "Stage 6" +msgstr "阶段 6" + +#: ../../chip_adaptation_guide/edge/08_tools_and_skills/skill_catalog.md:1 +msgid "Orchestrate the complete release pipeline." +msgstr "完整发布 pipeline 编排" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/advanced_quality_requirements.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/advanced_quality_requirements.mo new file mode 100644 index 0000000000..ecf9109ccd Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/advanced_quality_requirements.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/advanced_quality_requirements.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/advanced_quality_requirements.po new file mode 100644 index 0000000000..2d5e735cea --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/advanced_quality_requirements.po @@ -0,0 +1,44 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:38+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/advanced_quality_requirements.md:2 +msgid "Advanced Cooperation: Deterministic Performance and Quality Requirements" +msgstr "高级合作的性能与质量确定性要求" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/advanced_quality_requirements.md:4 +msgid "" +"Performance metrics are prerequisites for advanced cooperation, Day 0 " +"plans, distribution validation, and showroom display; they must not be " +"treated as post-release promotional items." +msgstr "性能指标是进入高级合作、Day 0 计划、发行版验证和展厅展示的前置条件,不作为事后宣传项处理。" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/advanced_quality_requirements.md:6 +msgid "" +"Vendors must maintain the mainstream diff so that chip adaptation on the " +"mainline has continuous ownership." +msgstr "厂商承诺维护 Mainstream Diff,确保主线上的芯片适配有人持续跟进。" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/advanced_quality_requirements.md:8 +msgid "" +"FlagOS upgrades and vendor Driver/SDK upgrades must both enter CI/CD " +"validation." +msgstr "FlagOS 升级与厂商 Driver/SDK 升级共同进入 CI/CD 验证。" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/advanced_quality_requirements.md:10 +msgid "" +"Day 0 releases, distribution plans, and policy recommendation plans must " +"meet maturity, support, performance, and quality requirements." +msgstr "Day 0 发布、发行版或政策推荐计划时,需满足成熟度、支持力度、性能和质量要求。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/index.mo new file mode 100644 index 0000000000..d1de68f319 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/index.po new file mode 100644 index 0000000000..c7d5104b4f --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/index.po @@ -0,0 +1,19 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-20 00:00+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +msgid "Promotion Criteria Checklist" +msgstr "晋升条件清单" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_advanced.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_advanced.mo new file mode 100644 index 0000000000..99386f9dcd Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_advanced.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_advanced.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_advanced.po new file mode 100644 index 0000000000..3528ff87e6 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_advanced.po @@ -0,0 +1,59 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:38+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:2 +msgid "Promotion from Intermediate to Advanced" +msgstr "中级 → 高级的必要条件" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:4 +msgid "" +"If the hardware supports FlagCX, complete the adaptation and maintain it " +"continuously; the SVT two-machine cluster must remain stable." +msgstr "如果为支持 FlagCX 的硬件,需适配完成且持续维护,SVT 双机集群稳定" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:6 +msgid "Use the FlagTree compiler by default." +msgstr "默认启用 FlagTree 编译器" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:8 +msgid "Pass the FlagOS CTS, VTS, and STS certification tests." +msgstr "通过 FlagOS CTS + VTS + STS 认证测试" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:10 +msgid "" +"Establish a compatibility matrix for core dependencies such as PyTorch " +"and Triton." +msgstr "建立核心依赖(PyTorch、Triton)兼容性矩阵" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:12 +msgid "" +"Complete the DevSecOps system and close the loop for security incident " +"response." +msgstr "完成 DevSecOps 体系,实现安全事件响应闭环" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:14 +msgid "Co-build the FlagGems operator library and contribute code." +msgstr "参与 FlagGems 算子库共建并贡献代码" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:16 +msgid "" +"Prepare the dedicated FlagRelease machine and continuously publish model " +"images." +msgstr "FlagRelease 专机就绪,模型镜像持续发布" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_advanced.md:18 +msgid "Work with operating-system distributions and cloud partner communities." +msgstr "协作推进,带入操作系统发行版和云合作社区" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_intermediate.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_intermediate.mo new file mode 100644 index 0000000000..fd71c335ab Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_intermediate.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_intermediate.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_intermediate.po new file mode 100644 index 0000000000..0fb0f9344d --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/09_promotion_criteria/promotion_to_intermediate.po @@ -0,0 +1,50 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-21 19:38+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:2 +msgid "Requirements for promotion from basic to intermediate" +msgstr "初级到中级的必要条件" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:4 +msgid "The FlagTree chip branch runs stably for at least four weeks" +msgstr "FlagTree 芯片分支稳定运行 ≥4 周" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:6 +msgid "FlagGems operator test coverage is at least 80%" +msgstr "FlagGems 算子测试覆盖 ≥80%" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:8 +msgid "" +"The CI single machine is in the FlagOS resource pool with availability of" +" at least 99%" +msgstr "CI 单机已纳入 FlagOS 资源池,可用性 ≥99%" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:10 +# +msgid "The chip passes Sig repository functional validation." +msgstr "芯片通过 Sig 仓功能验证" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:12 +msgid "Establish an SBOM scanning mechanism" +msgstr "建立 SBOM 扫描机制" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:14 +msgid "The driver/SDK can be obtained unattended" +msgstr "驱动/SDK 可无人值守获取" + +#: ../../chip_adaptation_guide/_shared/09_promotion_criteria/promotion_to_intermediate.md:16 +msgid "Sign and pass legal compliance review" +msgstr "签署并通过法务合规审查" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/10_risks_and_maintenance/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/10_risks_and_maintenance/index.mo new file mode 100644 index 0000000000..c605a651a4 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/10_risks_and_maintenance/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/10_risks_and_maintenance/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/10_risks_and_maintenance/index.po new file mode 100644 index 0000000000..0e53358c09 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge/10_risks_and_maintenance/index.po @@ -0,0 +1,101 @@ +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-22 18:56+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/_shared/10_risks_and_maintenance/index.md:2 +msgid "Risks and Maintenance" +msgstr "风险与注意事项" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "**Risk or requirement**" +msgstr "**风险项**" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "**Description**" +msgstr "**说明**" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "Driver access method" +msgstr "驱动获取方式" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "" +"Refreshing token authentication every 30 minutes is impractical for " +"automation. Driver and SDK packages should be openly downloadable or use " +"long-lived authorization." +msgstr "出于自动化考虑,每 30 分钟刷新一次 token 认证的方式不太现实,建议驱动/SDK 开放下载或使用长期有效授权。" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "Specific dependency trade-offs" +msgstr "特定依赖折衷" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "" +"Some trade-offs are allowed at the intermediate stage, but specific " +"dependencies must be eliminated before advanced cooperation." +msgstr "中级阶段允许保留部分折衷方案,但进入高级合作前必须消除特定依赖。" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "Image build neutrality" +msgstr "镜像构建中立性" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "" +"Use neutral base images to demonstrate compatibility rather than relying " +"entirely on vendor-customized images." +msgstr "建议使用中立基础镜像体现兼容性,而非完全依赖厂商定制镜像。" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "Long-term maintenance commitment" +msgstr "长期维护承诺" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "" +"Advanced cooperation requires continuous security maintenance and " +"progressive improvement of TLS and other version lifecycle management." +msgstr "高级合作要求对安全性问题承诺持续维护,并逐步完善 TLS 等版本生命周期管理。" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "Heterogeneous mixed efficiency" +msgstr "异构混合效率" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "" +"When a chip is used in a heterogeneous cluster at the intermediate stage," +" pay particular attention to mixed communication efficiency of at least " +"81%." +msgstr "中级阶段若芯片用于异构集群,需额外关注混合通信效率 ≥81%。" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "Device SLA" +msgstr "设备 SLA" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "" +"CI devices must stay online 24/7, SVT machines must be isolated, and the " +"FR machine must be configured independently. Long-term offline status may" +" cause certification downgrade or revocation." +msgstr "CI 设备要求 7×24 在线,SVT 隔离部署,FR 专机独立配置。长期离线可能导致认证降级或撤销。" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "Waiver period" +msgstr "豁免期限" + +#: ../../chip_adaptation_guide/edge/10_risks_and_maintenance/index.md:1 +msgid "" +"Each certification-test waiver may last no more than six months. An " +"unresolved waiver must be resubmitted or may result in certification " +"downgrade." +msgstr "认证测试中的每项豁免最长 6 个月,到期未修复需重新申请或接受认证降级。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge_adaptation_guide_index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge_adaptation_guide_index.mo new file mode 100644 index 0000000000..8e10e9ff70 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge_adaptation_guide_index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge_adaptation_guide_index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge_adaptation_guide_index.po new file mode 100644 index 0000000000..46b4dc3318 --- /dev/null +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/chip_adaptation_guide/edge_adaptation_guide_index.po @@ -0,0 +1,27 @@ +# FlagOS chip adaptation guide Chinese translation. +# Generated from English POT and preserved Chinese baseline. +msgid "" +msgstr "" +"Project-Id-Version: FlagOS Documentation \n" +"Report-Msgid-Bugs-To: \n" +"POT-Creation-Date: 2026-09-20 00:00+0800\n" +"PO-Revision-Date: 2026-09-20 00:00+0800\n" +"Last-Translator: FlagOS Community \n" +"Language: zh_CN\n" +"Language-Team: zh_CN \n" +"Plural-Forms: nplurals=1; plural=0;\n" +"MIME-Version: 1.0\n" +"Content-Type: text/plain; charset=utf-8\n" +"Content-Transfer-Encoding: 8bit\n" +"Generated-By: Babel 2.17.0\n" + +#: ../../chip_adaptation_guide/edge_adaptation_guide_index.md:3 +msgid "FlagOS Southbound Chip Adaptation Guide: Edge Chips" +msgstr "FlagOS 南向芯片适配指南:端侧芯片" + +#: ../../chip_adaptation_guide/edge_adaptation_guide_index.md:5 +msgid "This guide is for edge AI chip vendors, FlagOS adaptation engineers, system software developers, and distribution partners. It describes the complete path for integrating edge chips with FlagOS, including cooperation levels, device investment, the base environment, the FlagTree compiler, FlagGems operators, compatibility certification, distribution validation, quality metrics, automation tools, promotion criteria, and risk management." +msgstr "本手册面向端侧 AI 芯片厂商、FlagOS 适配工程师、系统软件开发者和发行版合作伙伴,介绍端侧芯片接入 FlagOS 的完整路径。内容覆盖合作分级、设备投入、基础环境、FlagTree 编译器、FlagGems 算子、兼容性认证、发行版验证、质量指标、自动化工具、晋升条件和风险管理。" + +msgid "Edge adaptation focuses on stable operation on single-node or small-scale devices, resource-constrained environments, driver and firmware coordination, and end-to-end efficiency. Edge adaptation does not require the FlagCX communication library or the FlagScale large-model training and inference framework by default; if a product explicitly supports these capabilities, they can be evaluated separately within the actual cooperation scope." +msgstr "端侧适配重点关注单机或小规模设备上的稳定运行、资源受限环境、驱动固件协同和端到端效率。端侧适配默认不要求 FlagCX 通信库和 FlagScale 大模型训练推理框架;如具体产品明确支持相关能力,可按实际合作范围另行评估。" diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index-backup.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index-backup.mo new file mode 100644 index 0000000000..0e15f7a0ae Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index-backup.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index-backup.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index-backup.po index 970e113a20..f09b4c1b57 100644 --- a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index-backup.po +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index-backup.po @@ -23,4 +23,3 @@ msgstr "" #: ../../flagos_homepage_en/index-backup.md:5 msgid "Documentation" msgstr "" - diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index.mo new file mode 100644 index 0000000000..644a1fbe69 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index.po index 215b165fa6..d99383e88d 100644 --- a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index.po +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/index.po @@ -8,7 +8,7 @@ msgid "" msgstr "" "Project-Id-Version: FlagOS Documentation \n" "Report-Msgid-Bugs-To: \n" -"POT-Creation-Date: 2026-06-25 13:35+0800\n" +"POT-Creation-Date: 2026-09-20 17:05+0800\n" "PO-Revision-Date: 2026-06-23 15:10+0800\n" "Last-Translator: FlagOS Community \n" "Language: zh_CN\n" @@ -19,80 +19,88 @@ msgstr "" "Content-Transfer-Encoding: 8bit\n" "Generated-By: Babel 2.17.0\n" -#: ../../flagos_homepage/index.md:5 +#: ../../index.md:5 msgid "Documentation" msgstr "文档中心" -#: ../../flagos_homepage/index.md:10 +#: ../../index.md:10 msgid "FlagOS" msgstr "FlagOS" -#: ../../flagos_homepage/index.md:12 +#: ../../index.md:12 msgid "" "A unified, open-source system software stack designed for a variety of AI" " chips" msgstr "一个统一的开源系统软件栈,专为多种 AI 芯片设计" -#: ../../flagos_homepage/index.md:14 +#: ../../index.md:14 #, python-brace-format -msgid "[FlagOS Overview](overview.md){ .flagos-outline-btn }" -msgstr "[FlagOS 概览](overview.md){ .flagos-outline-btn }" +msgid "" +"[FlagOS Overview](overview.md){ .flagos-outline-btn } [Cloud Chip " +"Adaptation Guide](chip_adaptation_guide/cloud_adaptation_guide_index.md){" +" .flagos-outline-btn } [Edge Chip Adaptation " +"Guide](chip_adaptation_guide/edge_adaptation_guide_index.md){ .flagos-" +"outline-btn }" +msgstr "" +"[FlagOS 概览](overview.md){ .flagos-outline-btn } [云端芯片适配指南](chip_adaptation_guide/cloud_adaptation_guide_index.md){" +" .flagos-outline-btn } [端侧芯片适配指南](chip_adaptation_guide/edge_adaptation_guide_index.md){ .flagos-" +"outline-btn }" -#: ../../flagos_homepage/index.md:17 +#: ../../index.md:27 msgid "FlagOS Core Libraries" msgstr "FlagOS 核心库" -#: ../../flagos_homepage/index.md:23 +#: ../../index.md:33 msgid "Operator Libraries" msgstr "算子库" -#: ../../flagos_homepage/index.md:26 +#: ../../index.md:36 msgid "" "High-performance operator libraries optimized for diverse hardware " "backends." msgstr "面向多种硬件后端优化的高性能算子库。" -#: ../../flagos_homepage/index.md:30 +#: ../../index.md:40 msgid "**General-Purpose Operator Library**" msgstr "**通用算子库**" -#: ../../flagos_homepage/index.md:32 +#: ../../index.md:42 msgid "**FlagGems**" msgstr "**FlagGems**" -#: ../../flagos_homepage/index.md:34 +#: ../../index.md:44 msgid "Triton-based general-purpose operator library." msgstr "基于 Triton 的通用算子库。" -#: ../../flagos_homepage/index.md:36 +#: ../../index.md:46 msgid "" "[View Documentation " "→](https://docs.flagos.io/projects/FlagGems/en/latest/)" msgstr "[查看文档 →](https://docs.flagos.io/projects/FlagGems/zh-cn/latest/)" -#: ../../flagos_homepage/index.md:40 +#: ../../index.md:50 msgid "**Fused Operator Libraries**" msgstr "**融合算子库**" -#: ../../flagos_homepage/index.md:42 +#: ../../index.md:52 msgid "**FlagGems-vllm**" msgstr "**FlagGems-vllm**" -#: ../../flagos_homepage/index.md:44 +#: ../../index.md:54 msgid "Optimized vLLM operators for multiple backends." msgstr "面向多种后端优化的 vLLM 算子。" -#: ../../flagos_homepage/index.md:46 +#: ../../index.md:56 msgid "" "[View Documentation →](https://docs.flagos.io/projects/FlagGems-" "vllm/en/latest/)" msgstr "[查看文档 →](https://docs.flagos.io/projects/FlagGems-vllm/zh-cn/latest/)" -#: ../../flagos_homepage/index.md:50 +#: ../../index.md:60 msgid "**Multi-Domain Operator Libraries**" msgstr "**多领域算子库**" -#: ../../flagos_homepage/index.md:52 +#: ../../index.md:62 msgid "" "**FlagDNN** — Deep learning operators. [View Documentation " "→](https://docs.flagos.io/projects/FlagDNN/en/latest/)" @@ -100,7 +108,7 @@ msgstr "" "**FlagDNN** — 深度学习算子。[查看文档 →](https://docs.flagos.io/projects/FlagDNN/zh-" "cn/latest/)" -#: ../../flagos_homepage/index.md:53 +#: ../../index.md:63 msgid "" "**FlagBLAS** — BLAS numerical library. [View Documentation " "→](https://docs.flagos.io/projects/FlagBLAS/en/latest/)" @@ -108,7 +116,7 @@ msgstr "" "**FlagBLAS** — BLAS 数值库。[查看文档 →](https://docs.flagos.io/projects/FlagBLAS" "/zh-cn/latest/)" -#: ../../flagos_homepage/index.md:54 +#: ../../index.md:64 msgid "" "**FlagFFT** — GPU FFT library. [View Documentation " "→](https://docs.flagos.io/projects/FlagFFT/en/latest/)" @@ -116,7 +124,7 @@ msgstr "" "**FlagFFT** — GPU FFT 库。[查看文档 →](https://docs.flagos.io/projects/FlagFFT" "/zh-cn/latest/)" -#: ../../flagos_homepage/index.md:55 +#: ../../index.md:65 msgid "" "**FlagSparse** — Sparse computation. [View Documentation " "→](https://docs.flagos.io/projects/FlagSparse/en/latest/)" @@ -124,7 +132,7 @@ msgstr "" "**FlagSparse** — 稀疏计算。[查看文档 →](https://docs.flagos.io/projects/FlagSparse" "/zh-cn/latest/)" -#: ../../flagos_homepage/index.md:56 +#: ../../index.md:66 msgid "" "**FlagTensor** — Tensor primitives. [View Documentation " "→](https://docs.flagos.io/projects/FlagTensor/en/latest/)" @@ -132,7 +140,7 @@ msgstr "" "**FlagTensor** — 张量原语。[查看文档 →](https://docs.flagos.io/projects/FlagTensor" "/zh-cn/latest/)" -#: ../../flagos_homepage/index.md:57 +#: ../../index.md:67 msgid "" "**FlagAudio** — Audio processing. [View Documentation " "→](https://docs.flagos.io/projects/FlagAudio/en/latest/)" @@ -140,21 +148,21 @@ msgstr "" "**FlagAudio** — 音频处理。[查看文档 →](https://docs.flagos.io/projects/FlagAudio" "/zh-cn/latest/)" -#: ../../flagos_homepage/index.md:66 +#: ../../index.md:76 msgid "Compiler" msgstr "编译器" -#: ../../flagos_homepage/index.md:69 +#: ../../index.md:79 msgid "**FlagTree**" msgstr "**FlagTree**" -#: ../../flagos_homepage/index.md:71 +#: ../../index.md:81 msgid "" "An open-source, unified compiler for multiple AI chips, advancing and " "expanding the Triton ecosystem across diverse hardware platforms." msgstr "面向多种 AI 芯片的开源统一编译器,在多种硬件平台上推进和扩展 Triton 生态系统。" -#: ../../flagos_homepage/index.md:74 +#: ../../index.md:84 #, python-brace-format msgid "" "[View Documentation " @@ -163,21 +171,21 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/FlagTree/zh-cn/latest/){ .card-" "link-sd }" -#: ../../flagos_homepage/index.md:77 +#: ../../index.md:87 msgid "Training & Inference Framework" msgstr "训练与推理框架" -#: ../../flagos_homepage/index.md:80 +#: ../../index.md:90 msgid "**FlagScale**" msgstr "**FlagScale**" -#: ../../flagos_homepage/index.md:82 +#: ../../index.md:92 msgid "" "A comprehensive toolkit designed to support the entire lifecycle of large" " models, from training to inference and deployment." msgstr "全面支持大模型全生命周期的工具集,涵盖从训练到推理和部署。" -#: ../../flagos_homepage/index.md:85 +#: ../../index.md:95 #, python-brace-format msgid "" "[View Documentation " @@ -186,22 +194,22 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/FlagScale/zh-cn/latest/){ .card-" "link-sd }" -#: ../../flagos_homepage/index.md:88 +#: ../../index.md:98 msgid "Communication Library" msgstr "通信库" -#: ../../flagos_homepage/index.md:91 +#: ../../index.md:101 msgid "**FlagCX**" msgstr "**FlagCX**" -#: ../../flagos_homepage/index.md:93 +#: ../../index.md:103 msgid "" "A scalable and adaptive unified communication library for cross-chip " "environments, delivering high-performance collective communication " "capabilities." msgstr "面向跨芯片环境的可扩展自适应统一通信库,提供高性能集合通信能力。" -#: ../../flagos_homepage/index.md:96 +#: ../../index.md:106 #, python-brace-format msgid "" "[View Documentation " @@ -210,21 +218,21 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/FlagCX/zh-cn/latest/){ .card-" "link-sd }" -#: ../../flagos_homepage/index.md:102 +#: ../../index.md:112 msgid "FlagOS Plugins for Diverse Chips" msgstr "FlagOS 多芯片插件" -#: ../../flagos_homepage/index.md:108 +#: ../../index.md:118 msgid "vllm-plugin-FL" msgstr "vllm-plugin-FL" -#: ../../flagos_homepage/index.md:111 +#: ../../index.md:121 msgid "" "A plugin for the vLLM inference/serving framework, built on FlagOS's " "unified multi-chip backend." msgstr "vLLM 推理/服务框架插件,基于 FlagOS 统一多芯片后端构建。" -#: ../../flagos_homepage/index.md:114 +#: ../../index.md:124 #, python-brace-format msgid "" "[View Documentation →](https://docs.flagos.io/projects/vllm-plugin-" @@ -233,17 +241,17 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/vllm-plugin-FL/zh-cn/latest/){ " ".card-link-sd }" -#: ../../flagos_homepage/index.md:117 +#: ../../index.md:127 msgid "Megatron-LM-FL" msgstr "Megatron-LM-FL" -#: ../../flagos_homepage/index.md:120 +#: ../../index.md:130 msgid "" "A fork of Megatron-LM that introduces a plugin-based architecture for " "supporting diverse AI chips, built on top of FlagOS." msgstr "Megatron-LM 的分支,引入插件化架构以支持多种 AI 芯片,基于 FlagOS 构建。" -#: ../../flagos_homepage/index.md:123 +#: ../../index.md:133 #, python-brace-format msgid "" "[View Documentation →](https://docs.flagos.io/projects/Megatron-LM-" @@ -252,17 +260,17 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/Megatron-LM-FL/zh-cn/latest/){ " ".card-link-sd }" -#: ../../flagos_homepage/index.md:126 +#: ../../index.md:136 msgid "TransformerEngine-FL" msgstr "TransformerEngine-FL" -#: ../../flagos_homepage/index.md:129 +#: ../../index.md:139 msgid "" "A fork of TransformerEngine that introduces a plugin-based architecture " "for supporting diverse AI chips, built on top of FlagOS." msgstr "TransformerEngine 的分支,引入插件化架构以支持多种 AI 芯片,基于 FlagOS 构建。" -#: ../../flagos_homepage/index.md:132 +#: ../../index.md:142 #, python-brace-format msgid "" "[View Documentation →](https://docs.flagos.io/projects/TransformerEngine-" @@ -271,17 +279,17 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/TransformerEngine-FL/zh-" "cn/latest/){ .card-link-sd }" -#: ../../flagos_homepage/index.md:135 +#: ../../index.md:145 msgid "verl-FL" msgstr "verl-FL" -#: ../../flagos_homepage/index.md:138 +#: ../../index.md:148 msgid "" "A fork of veRL that extends the upstream library with multi-chip/multi-" "hardware support via the FlagOS ecosystem." msgstr "veRL 的分支,通过 FlagOS 生态系统扩展上游库的多芯片/多硬件支持。" -#: ../../flagos_homepage/index.md:141 +#: ../../index.md:151 #, python-brace-format msgid "" "[View Documentation →](https://docs.flagos.io/projects/verl-" @@ -290,11 +298,11 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/verl-FL/zh-cn/latest/){ .card-" "link-sd }" -#: ../../flagos_homepage/index.md:144 +#: ../../index.md:154 msgid "PyTorch-Plugin-FL" msgstr "PyTorch-Plugin-FL" -#: ../../flagos_homepage/index.md:147 +#: ../../index.md:157 msgid "" "A custom PyTorch device plugin based on the PrivateUse1 extension " "mechanism, registering FlagGems high-performance Triton operators as the " @@ -303,7 +311,7 @@ msgstr "" "基于 PrivateUse1 扩展机制的自定义 PyTorch 设备插件,将 FlagGems 高性能 Triton 算子注册为 flagos " "设备后端。" -#: ../../flagos_homepage/index.md:150 +#: ../../index.md:160 #, python-brace-format msgid "" "[View Documentation →](https://docs.flagos.io/projects/PyTorch-Plugin-" @@ -312,18 +320,18 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/PyTorch-Plugin-FL/zh-" "cn/latest/){ .card-link-sd }" -#: ../../flagos_homepage/index.md:153 +#: ../../index.md:163 msgid "sglang-plugin-FL" msgstr "sglang-plugin-FL" -#: ../../flagos_homepage/index.md:156 +#: ../../index.md:166 msgid "" "An out-of-tree (OOT) plugin for SGLang, built on FlagOS's unified multi-" "chip backend, extending SGLang's inference capabilities across diverse " "hardware platforms." msgstr "SGLang 的树外 (OOT) 插件,基于 FlagOS 统一多芯片后端构建,将 SGLang 的推理能力扩展到多种硬件平台。" -#: ../../flagos_homepage/index.md:159 +#: ../../index.md:169 #, python-brace-format msgid "" "[View Documentation →](https://docs.flagos.io/projects/sglang-plugin-" @@ -332,21 +340,21 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/sglang-plugin-FL/zh-cn/latest/){" " .card-link-sd }" -#: ../../flagos_homepage/index.md:165 +#: ../../index.md:175 msgid "FlagOS Domain-Specific Projects" msgstr "FlagOS 领域专用项目" -#: ../../flagos_homepage/index.md:171 +#: ../../index.md:181 msgid "FlagOS-Robo" msgstr "FlagOS-Robo" -#: ../../flagos_homepage/index.md:174 +#: ../../index.md:184 msgid "" "An integrated training and inference framework for AI models used in " "robots, so-called Embodied Intelligence." msgstr "面向机器人 AI 模型的集成训练与推理框架,即具身智能。" -#: ../../flagos_homepage/index.md:177 +#: ../../index.md:187 #, python-brace-format msgid "" "[View Documentation →](https://docs.flagos.io/projects/FlagOS-" @@ -355,17 +363,17 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/FlagOS-Robo/zh-cn/latest/){ " ".card-link-sd }" -#: ../../flagos_homepage/index.md:180 +#: ../../index.md:190 msgid "FlagQuantum" msgstr "FlagQuantum" -#: ../../flagos_homepage/index.md:183 +#: ../../index.md:193 msgid "" "A high-performance distributed quantum statevector simulator built on " "PyTorch, enabling quantum circuit simulation across multiple GPUs." msgstr "基于 PyTorch 构建的高性能分布式量子态矢量模拟器,支持跨多 GPU 的量子电路模拟。" -#: ../../flagos_homepage/index.md:186 +#: ../../index.md:196 #, python-brace-format msgid "" "[View Documentation " @@ -375,19 +383,19 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/FlagQuantum/zh-cn/latest/){ " ".card-link-sd }" -#: ../../flagos_homepage/index.md:192 +#: ../../index.md:202 msgid "FlagOS Developer Tools" msgstr "FlagOS 开发者工具" -#: ../../flagos_homepage/index.md:198 +#: ../../index.md:208 msgid "KernelGen" msgstr "KernelGen" -#: ../../flagos_homepage/index.md:201 +#: ../../index.md:211 msgid "An operator auto-generation tool." msgstr "算子自动生成工具。" -#: ../../flagos_homepage/index.md:204 +#: ../../index.md:214 #, python-brace-format msgid "" "[View Documentation " @@ -396,17 +404,17 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/kernelgen/zh-cn/latest/){ .card-" "link-sd }" -#: ../../flagos_homepage/index.md:207 +#: ../../index.md:217 msgid "KernelGenBench" msgstr "KernelGenBench" -#: ../../flagos_homepage/index.md:210 +#: ../../index.md:220 msgid "" "A benchmark framework for evaluating LLM and agent-based Triton kernel " "generation across multiple hardware platforms." msgstr "用于评估跨多硬件平台 LLM 和 Agent 驱动的 Triton kernel 生成的基准测试框架。" -#: ../../flagos_homepage/index.md:213 +#: ../../index.md:223 #, python-brace-format msgid "" "[View Documentation " @@ -416,32 +424,32 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/kernelgenbench/zh-cn/latest/){ " ".card-link-sd }" -#: ../../flagos_homepage/index.md:216 +#: ../../index.md:226 msgid "FlagOS Skills" msgstr "FlagOS Skills" -#: ../../flagos_homepage/index.md:220 +#: ../../index.md:230 msgid "" "Compatible with Claude Code, Cursor, Codex, and any agent supporting the " "Agent Skills standard." msgstr "兼容 Claude Code、Cursor、Codex 以及任何支持 Agent Skills 标准的 Agent。" -#: ../../flagos_homepage/index.md:223 +#: ../../index.md:233 #, python-brace-format msgid "" "[View Documentation →](https://github.com/flagos-ai/skills){ .card-link-" "sd }" msgstr "[查看文档 →](https://github.com/flagos-ai/skills){ .card-link-sd }" -#: ../../flagos_homepage/index.md:226 +#: ../../index.md:236 msgid "Online Laboratory" msgstr "线上实验室" -#: ../../flagos_homepage/index.md:229 +#: ../../index.md:239 msgid "An online laboratory providing cloud-based development environments." msgstr "提供云端开发环境的线上实验室。" -#: ../../flagos_homepage/index.md:232 +#: ../../index.md:242 #, python-brace-format msgid "" "[View Documentation " @@ -471,17 +479,17 @@ msgstr "" msgid "FlagOS Platform Services" msgstr "FlagOS 平台服务" -#: ../../flagos_homepage/index.md:244 +#: ../../index.md:254 msgid "FlagRelease" msgstr "FlagRelease" -#: ../../flagos_homepage/index.md:247 +#: ../../index.md:257 msgid "" "An automated platform for the cross-chip migration and release of open-" "source large models." msgstr "开源大模型跨芯片迁移与发布的自动化平台。" -#: ../../flagos_homepage/index.md:250 +#: ../../index.md:260 #, python-brace-format msgid "" "[View Documentation " @@ -491,15 +499,15 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/FlagRelease/zh-cn/latest/){ " ".card-link-sd }" -#: ../../flagos_homepage/index.md:253 +#: ../../index.md:263 msgid "FlagPerf" msgstr "FlagPerf" -#: ../../flagos_homepage/index.md:256 +#: ../../index.md:266 msgid "An integrated AI hardware evaluation engine." msgstr "集成式 AI 硬件评测引擎。" -#: ../../flagos_homepage/index.md:259 +#: ../../index.md:269 #, python-brace-format msgid "" "[View Documentation " @@ -508,17 +516,17 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/FlagPerf/zh-cn/latest/){ .card-" "link-sd }" -#: ../../flagos_homepage/index.md:262 +#: ../../index.md:272 msgid "FlagCICD" msgstr "FlagCICD" -#: ../../flagos_homepage/index.md:266 +#: ../../index.md:276 msgid "" "A CI/CD toolchain that streamlines large-model development across diverse" " AI chips." msgstr "简化跨多种 AI 芯片大模型开发的 CI/CD 工具链。" -#: ../../flagos_homepage/index.md:269 +#: ../../index.md:279 #, python-brace-format msgid "" "[View Documentation →](https://docs.flagos.io/projects/FlagCICD/zh-" @@ -527,28 +535,15 @@ msgstr "" "[查看文档 →](https://docs.flagos.io/projects/FlagCICD/zh-cn/latest/){ .card-" "link-sd }" -#: ../../flagos_homepage/index.md:278 +#: ../../index.md:288 msgid "Start to Use FlagOS" msgstr "开始使用 FlagOS" -#: ../../flagos_homepage/index.md:280 +#: ../../index.md:290 msgid "Join us to co-build an open AI chip development ecosystem" msgstr "加入我们,共建开放 AI 芯片开发生态" -#: ../../flagos_homepage/index.md:282 +#: ../../index.md:292 #, python-brace-format msgid "[FlagOS Homepage](https://flagos.io/){ .btn .btn-primary .btn-lg }" msgstr "[FlagOS 主页](https://flagos.io/){ .btn .btn-primary .btn-lg }" - -#~ msgid "**Fused**" -#~ msgstr "**融合**" - -#~ msgid "**Multi-Domain**" -#~ msgstr "**多领域**" - -#~ msgid "FlagDNN • FlagBLAS • FlagFFT • FlagSparse • FlagTensor • FlagAudio" -#~ msgstr "FlagDNN • FlagBLAS • FlagFFT • FlagSparse • FlagTensor • FlagAudio" - -#~ msgid "[View all →](https://docs.flagos.io/projects/FlagDNN/en/latest/)" -#~ msgstr "[查看全部 →](https://docs.flagos.io/projects/FlagDNN/en/latest/)" - diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/overview.mo b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/overview.mo new file mode 100644 index 0000000000..a6eef9a046 Binary files /dev/null and b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/overview.mo differ diff --git a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/overview.po b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/overview.po index 60d17e2265..84b7cd496a 100644 --- a/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/overview.po +++ b/docs/flagos_homepage/locale/zh_CN/LC_MESSAGES/overview.po @@ -8,7 +8,7 @@ msgid "" msgstr "" "Project-Id-Version: FlagOS Documentation \n" "Report-Msgid-Bugs-To: \n" -"POT-Creation-Date: 2026-06-24 13:26+0800\n" +"POT-Creation-Date: 2026-09-20 17:05+0800\n" "PO-Revision-Date: 2026-06-22 13:15+0800\n" "Last-Translator: FlagOS Community \n" "Language: zh_CN\n" @@ -19,62 +19,54 @@ msgstr "" "Content-Transfer-Encoding: 8bit\n" "Generated-By: Babel 2.17.0\n" -#: ../../flagos_homepage/overview.md:1 +#: ../../overview.md:1 msgid "FlagOS Overview" msgstr "FlagOS 概览" -#: ../../flagos_homepage/overview.md:3 +#: ../../overview.md:3 msgid "" "FlagOS is a fully open-source AI system software stack for heterogeneous " "AI chips, allowing AI models to be developed once and seamlessly ported " "to a wide range of AI hardware with minimal effort." msgstr "FlagOS 是一个完全开源的异构 AI 芯片系统软件栈,允许 AI 模型一次开发即可无缝移植到广泛的 AI 硬件平台,实现最小化的适配成本。" -#: ../../flagos_homepage/overview.md:6 -msgid "" -"This is a preview release. The version number shown is a pre-release " -"identifier and may change upon final release. Content in this preview is " -"for reference only and does not constitute a commitment or warranty for " -"the final product." -msgstr "这是一个预览版本。显示的版本号为预发布标识符,最终发布时可能发生变化。本预览版内容仅供参考,不构成对最终产品的承诺或保证。" - -#: ../../flagos_homepage/overview.md:9 ../../flagos_homepage/overview.md:13 +#: ../../overview.md:6 ../../overview.md:10 msgid "FlagOS architecture" msgstr "FlagOS 架构" -#: ../../flagos_homepage/overview.md:11 +#: ../../overview.md:8 msgid "" "The figure below shows the position of FlagOS in the AI ecosystem and its" " composition modules." msgstr "下图展示了 FlagOS 在 AI 生态系统中的位置及其组成模块。" -#: ../../flagos_homepage/overview.md:13 +#: ../../overview.md:10 msgid "![FlagOS architecture](images/flagos-architecture-en.png)" msgstr "![FlagOS 架构](images/flagos-architecture-zh.png)" -#: ../../flagos_homepage/overview.md:15 +#: ../../overview.md:12 msgid "" "FlagOS 2.1 comprises the following core libraries, plugins, domain-" "specific projects, developer tools, and platform services." msgstr "FlagOS 2.1 包含以下核心库、插件、领域专用项目、开发者工具和平台服务。" -#: ../../flagos_homepage/overview.md:17 +#: ../../overview.md:14 msgid "Open-source core libraries" msgstr "开源核心库" -#: ../../flagos_homepage/overview.md:19 +#: ../../overview.md:16 msgid "Operator libraries" msgstr "算子库" -#: ../../flagos_homepage/overview.md:20 +#: ../../overview.md:17 msgid "**General-purpose operator library**" msgstr "**通用算子库**" -#: ../../flagos_homepage/overview.md:21 +#: ../../overview.md:18 msgid "**FlagGems** (v5.3.0)" msgstr "**FlagGems** (v5.3.0)" -#: ../../flagos_homepage/overview.md:23 +#: ../../overview.md:20 msgid "" "FlagGems is a high-performance general-purpose operator library " "implemented with the Triton programming language and its extended " @@ -85,15 +77,15 @@ msgstr "" "FlagGems 是一个使用 Triton 编程语言及其扩展语言实现的高性能通用算子库。FlagGems " "旨在为大模型提供一套通用算子,加速多后端平台上的模型推理和训练。" -#: ../../flagos_homepage/overview.md:25 +#: ../../overview.md:22 msgid "**Fused operator libraries**" msgstr "**融合算子库**" -#: ../../flagos_homepage/overview.md:26 +#: ../../overview.md:23 msgid "**FlagGems-vllm** (v0.1.0)" msgstr "**FlagGems-vllm** (v0.1.0)" -#: ../../flagos_homepage/overview.md:28 +#: ../../overview.md:25 msgid "" "A high-performance operator library designed for multiple hardware " "backends. It provides optimized implementations of common vLLM operators " @@ -101,37 +93,37 @@ msgid "" "widely used models." msgstr "一个面向多硬件后端的高性能算子库。它提供常见 vLLM 算子的优化实现,支持多种广泛使用模型的高性能推理和部署。" -#: ../../flagos_homepage/overview.md:30 +#: ../../overview.md:27 msgid "**Multi-domain operator libraries**" msgstr "**多领域算子库**" -#: ../../flagos_homepage/overview.md:32 +#: ../../overview.md:29 msgid "**FlagDNN** (v0.2.0)" msgstr "**FlagDNN** (v0.2.0)" -#: ../../flagos_homepage/overview.md:34 +#: ../../overview.md:31 msgid "" "A deep neural network computing library oriented towards multiple chip " "backends. It provides high-performance implementations of common deep " "learning operators." msgstr "一个面向多芯片后端的深度神经网络计算库。它提供常见深度学习算子的高性能实现。" -#: ../../flagos_homepage/overview.md:36 +#: ../../overview.md:33 msgid "**FlagBLAS** (v0.2.0)" msgstr "**FlagBLAS** (v0.2.0)" -#: ../../flagos_homepage/overview.md:38 +#: ../../overview.md:35 msgid "" "A computing library that follows the BLAS standard interface and is " "oriented towards multiple chip backends. It defines core operations for " "numerical calculations." msgstr "一个遵循 BLAS 标准接口、面向多芯片后端的计算库。它定义数值计算的核心操作。" -#: ../../flagos_homepage/overview.md:40 +#: ../../overview.md:37 msgid "**FlagFFT** (v0.1.0)" msgstr "**FlagFFT** (v0.1.0)" -#: ../../flagos_homepage/overview.md:42 +#: ../../overview.md:39 msgid "" "A JIT-compiled GPU FFT library. It generates CUDA kernels at runtime via " "Triton/TLE and libtriton_jit, targeting arbitrary-length transforms that " @@ -140,21 +132,21 @@ msgstr "" "一个 JIT 编译的 GPU FFT 库。它通过 Triton/TLE 和 libtriton_jit 在运行时生成 CUDA 内核,针对 " "cuFFT 无法最优支持的任意长度变换。" -#: ../../flagos_homepage/overview.md:44 +#: ../../overview.md:41 msgid "**FlagSparse** (v0.2.0)" msgstr "**FlagSparse** (v0.2.0)" -#: ../../flagos_homepage/overview.md:46 +#: ../../overview.md:43 msgid "" "A domain-specific operator library that contains operators dedicated to " "sparse computation scenarios." msgstr "一个领域专用算子库,包含专门用于稀疏计算场景的算子。" -#: ../../flagos_homepage/overview.md:48 +#: ../../overview.md:45 msgid "**FlagTensor** (v0.2.0)" msgstr "**FlagTensor** (v0.2.0)" -#: ../../flagos_homepage/overview.md:50 +#: ../../overview.md:47 msgid "" "A high-performance tensor-primitive library implemented in Triton " "language. It provides optimized implementations of common tensor " @@ -164,22 +156,22 @@ msgstr "" "一个使用 Triton 语言实现的高性能张量原语库。它提供常见张量原语(一元、二元和张量缩并操作)的优化实现,并以 cuTensor " "为基准进行测试。" -#: ../../flagos_homepage/overview.md:52 +#: ../../overview.md:49 msgid "**FlagAudio** (v0.2.0)" msgstr "**FlagAudio** (v0.2.0)" -#: ../../flagos_homepage/overview.md:54 +#: ../../overview.md:51 msgid "" "A multi-backend computing library that adheres to Audio standard " "interfaces. It delivers a high-performance computing solution designed " "for audio signal processing and speech AI applications." msgstr "一个遵循 Audio 标准接口的多后端计算库。它为音频信号处理和语音 AI 应用提供高性能计算解决方案。" -#: ../../flagos_homepage/overview.md:56 +#: ../../overview.md:53 msgid "**FlagTree** (v0.6.0)" msgstr "**FlagTree** (v0.6.0)" -#: ../../flagos_homepage/overview.md:58 +#: ../../overview.md:55 msgid "" "FlagTree is an open-source, unified compiler for multiple AI chips. " "FlagTree is dedicated to building a compiler and associated tooling " @@ -195,11 +187,11 @@ msgstr "" "Triton 上下游生态系统,目标是支持现有适配方案、统一代码仓库,并实现从单一仓库快速支持多后端。对于上游模型用户,FlagTree " "提供跨多后端的统一编译支持;对于下游芯片厂商,FlagTree 提供集成到 Triton 生态的参考实现。" -#: ../../flagos_homepage/overview.md:60 +#: ../../overview.md:57 msgid "**FlagScale** (v2.0.0)" msgstr "**FlagScale** (v2.0.0)" -#: ../../flagos_homepage/overview.md:62 +#: ../../overview.md:59 msgid "" "FlagScale is a comprehensive toolkit designed to support the entire " "lifecycle of large models. FlagScale builds on the strengths of several " @@ -210,11 +202,11 @@ msgstr "" "FlagScale 是一个全面的大模型全生命周期工具集。FlagScale 基于 Megatron-LM 和 vLLM " "等多个知名开源项目的优势,为管理和扩展大模型提供稳健的端到端解决方案。" -#: ../../flagos_homepage/overview.md:64 +#: ../../overview.md:61 msgid "**FlagCX** (v0.13.0)" msgstr "**FlagCX** (v0.13.0)" -#: ../../flagos_homepage/overview.md:66 +#: ../../overview.md:63 msgid "" "FlagCX is a scalable and adaptive unified communication library for " "cross-chip environments. FlagCX delivers high-performance point-to-point " @@ -229,37 +221,35 @@ msgstr "" "为多芯片、多平台场景提供高性能的点对点和集合通信能力。通过利用每个平台的原生集合通信能力,FlagCX 采用设备缓冲区 IPC 和 RDMA " "等技术,在跨芯片和单芯片场景中实现高效的集合通信,同时提供通信优化的自适应调优能力。" -#: ../../flagos_homepage/overview.md:68 +#: ../../overview.md:65 msgid "Plugins" msgstr "插件" -#: ../../flagos_homepage/overview.md:70 +#: ../../overview.md:67 msgid "" "The FlagOS ecosystem enablement layer adopts a plugin architecture " "composed of the following modules. Each module bridges an upstream " "library and its backend engine with the FlagOS core libraries." msgstr "FlagOS 生态使能层采用插件架构,由以下模块组成。每个模块将上游库及其后端引擎与 FlagOS 核心库连接起来。" -#: ../../flagos_homepage/overview.md:72 +#: ../../overview.md:69 msgid "**vllm-plugin-FL** (v0.2.0)" msgstr "**vllm-plugin-FL** (v0.2.0)" -#: ../../flagos_homepage/overview.md:74 +#: ../../overview.md:71 msgid "" "vllm-plugin-FL extends the inference capabilities of vLLM to diverse AI " "chips, enabling efficient model serving beyond the original supported " -"hardware. Built on FlagOS's unified multi-chip backend — including the " -"unified operator library FlagGems and the unified communication library " -"FlagCX." +"hardware. Built on FlagOS's unified multi-chip backend." msgstr "" "vllm-plugin-FL 将 vLLM 的推理能力扩展到多种 AI 芯片,实现超越原始支持硬件的高效模型服务。基于 FlagOS " -"的统一多芯片后端构建——包括统一算子库 FlagGems 和统一通信库 FlagCX。" +"的统一多芯片后端构建。" -#: ../../flagos_homepage/overview.md:76 +#: ../../overview.md:73 msgid "**sglang-plugin-FL** (v0.1.0)" msgstr "**sglang-plugin-FL** (v0.1.0)" -#: ../../flagos_homepage/overview.md:78 +#: ../../overview.md:75 msgid "" "sglang-plugin-FL is an out-of-tree (OOT) plugin for SGLang, built on " "FlagOS's unified multi-chip backend. It extends SGLang's inference " @@ -268,11 +258,11 @@ msgstr "" "sglang-plugin-FL 是 SGLang 的一个树外 (OOT) 插件,基于 FlagOS 的统一多芯片后端构建。它将 SGLang " "的推理能力扩展到多种硬件平台。" -#: ../../flagos_homepage/overview.md:80 +#: ../../overview.md:77 msgid "**PyTorch-Plugin-FL** (v0.1.0)" msgstr "**PyTorch-Plugin-FL** (v0.1.0)" -#: ../../flagos_homepage/overview.md:82 +#: ../../overview.md:79 msgid "" "PyTorch-Plugin-FL is a custom PyTorch device plugin based on the " "PrivateUse1 extension mechanism, registering FlagGems high-performance " @@ -282,22 +272,22 @@ msgstr "" "PyTorch-Plugin-FL 是一个基于 PrivateUse1 扩展机制的自定义 PyTorch 设备插件,将 FlagGems 高性能 " "Triton 算子注册为 flagos 设备后端,实现统一的多芯片支持。" -#: ../../flagos_homepage/overview.md:84 +#: ../../overview.md:81 msgid "**Megatron-LM-FL** (v0.2.0)" msgstr "**Megatron-LM-FL** (v0.2.0)" -#: ../../flagos_homepage/overview.md:86 +#: ../../overview.md:83 msgid "" "Megatron-LM-FL extends the distributed training capabilities of Megatron-" "LM to diverse AI chips, supporting scalable large-model training across " "heterogeneous hardware." msgstr "Megatron-LM-FL 将 Megatron-LM 的分布式训练能力扩展到多种 AI 芯片,支持跨异构硬件的可扩展大模型训练。" -#: ../../flagos_homepage/overview.md:88 +#: ../../overview.md:85 msgid "**TransformerEngine-FL** (v0.2.0)" msgstr "**TransformerEngine-FL** (v0.2.0)" -#: ../../flagos_homepage/overview.md:90 +#: ../../overview.md:87 msgid "" "TransformerEngine-FL extends the transformer acceleration capabilities of" " Transformer Engine to diverse AI chips, enabling hardware-agnostic " @@ -306,18 +296,18 @@ msgstr "" "TransformerEngine-FL 将 Transformer Engine 的 transformer 加速能力扩展到多种 AI " "芯片,实现硬件无关的训练加速。" -#: ../../flagos_homepage/overview.md:92 +#: ../../overview.md:89 msgid "**verl-FL** (v0.2.0)" msgstr "**verl-FL** (v0.2.0)" -#: ../../flagos_homepage/overview.md:94 +#: ../../overview.md:91 msgid "" "verl-FL extends the reinforcement learning capabilities of veRL to " "diverse AI chips, broadening the hardware coverage for RL-based training " "workflows." msgstr "verl-FL 将 veRL 的强化学习能力扩展到多种 AI 芯片,拓宽基于强化学习训练工作流的硬件覆盖范围。" -#: ../../flagos_homepage/overview.md:96 +#: ../../overview.md:93 msgid "" "vllm-plugin-FL, Megatron-LM-FL, TransformerEngine-FL, and verl-FL can be " "used together with FlagScale. When only one or two capabilities are " @@ -326,19 +316,19 @@ msgid "" "backend engine with the relevant FlagOS core library modules, offering " "the flexibility to meet diverse user deployment scenarios." msgstr "" -"vllm-plugin-FL、Megatron-LM-FL、TransformerEngine-FL 和 verl-FL 可以与 FlagScale " -"配合使用。当只需要一两项能力(如训练、推理或强化学习)时,相应模块可以独立地将其上游库和后端引擎与相关 FlagOS " +"vllm-plugin-FL、Megatron-LM-FL、TransformerEngine-FL 和 verl-FL 可以与 " +"FlagScale 配合使用。当只需要一两项能力(如训练、推理或强化学习)时,相应模块可以独立地将其上游库和后端引擎与相关 FlagOS " "核心库模块连接,提供满足多样化用户部署场景的灵活性。" -#: ../../flagos_homepage/overview.md:98 +#: ../../overview.md:95 msgid "Domain-specific projects" msgstr "领域专用项目" -#: ../../flagos_homepage/overview.md:100 +#: ../../overview.md:97 msgid "**FlagOS-Robo** (v0.1.0)" msgstr "**FlagOS-Robo** (v0.1.0)" -#: ../../flagos_homepage/overview.md:102 +#: ../../overview.md:99 msgid "" "FlagOS-Robo is a chip-agnostic framework for training and deploying " "Vision Language Models (VLMs) and Vision Language Action (VLA) models " @@ -349,11 +339,11 @@ msgstr "" "FlagOS-Robo 是一个芯片无关的框架,用于在具身智能的端到云场景中训练和部署视觉语言模型 (VLM) 和视觉语言动作模型 (VLA)。它将" " VLM 视为任务规划的\"大脑\",将 VLA 模型视为生成机器人控制动作的\"小脑\"。" -#: ../../flagos_homepage/overview.md:104 +#: ../../overview.md:101 msgid "**FlagQuantum** (v0.1.0)" msgstr "**FlagQuantum** (v0.1.0)" -#: ../../flagos_homepage/overview.md:106 +#: ../../overview.md:103 msgid "" "FlagQuantum is a high-performance distributed quantum statevector " "simulator built on PyTorch, enabling quantum circuit simulation across " @@ -362,15 +352,15 @@ msgstr "" "FlagQuantum 是一个基于 PyTorch 构建的高性能分布式量子态矢量模拟器,支持跨多 GPU " "的量子电路模拟,具有自动分片和重新分片功能。" -#: ../../flagos_homepage/overview.md:108 +#: ../../overview.md:105 msgid "Developer tools" msgstr "开发者工具" -#: ../../flagos_homepage/overview.md:110 -msgid "**KernelGen** (v2.1)" -msgstr "**KernelGen** (v2.1)" +#: ../../overview.md:107 +msgid "**KernelGen** (v2.1.0)" +msgstr "**KernelGen** (v2.1.0)" -#: ../../flagos_homepage/overview.md:112 +#: ../../overview.md:109 msgid "" "KernelGen is an operator auto-generation tool. KernelGen is designed to " "construct operator definitions through natural language prompts, retrieve" @@ -381,11 +371,11 @@ msgstr "" "KernelGen 是一个算子自动生成工具。KernelGen " "旨在通过自然语言提示构建算子定义,检索现有相似算子定义,自动执行算子精度和性能测试,生成精度和性能测试结果,并产出 Triton Kernel。" -#: ../../flagos_homepage/overview.md:114 +#: ../../overview.md:111 msgid "**FlagOS Skills** (v1.1.0)" msgstr "**FlagOS Skills** (v1.1.0)" -#: ../../flagos_homepage/overview.md:116 +#: ../../overview.md:113 msgid "" "FlagOS Skills are agent-compatible capabilities designed to streamline " "key FlagOS workflows, including deployment, operator development, " @@ -395,25 +385,25 @@ msgstr "" "FlagOS Skills 是与 Agent 兼容的能力集,旨在简化关键 FlagOS 工作流程,包括部署、算子开发、迁移、适配和性能评估。兼容 " "Claude Code、Cursor、Codex 以及任何支持 Agent Skills 标准的 Agent。" -#: ../../flagos_homepage/overview.md:118 +#: ../../overview.md:115 msgid "**Online Laboratory**" msgstr "**线上实验室**" -#: ../../flagos_homepage/overview.md:120 +#: ../../overview.md:117 msgid "" "An online laboratory providing cloud-based development environments for " "FlagOS projects." msgstr "一个为 FlagOS 项目提供云端开发环境的线上实验室。" -#: ../../flagos_homepage/overview.md:122 +#: ../../overview.md:119 msgid "Platform services" msgstr "平台服务" -#: ../../flagos_homepage/overview.md:124 -msgid "**FlagRelease** (v0.2.0)" -msgstr "**FlagRelease** (v0.2.0)" +#: ../../overview.md:121 +msgid "**FlagRelease** (v0.1.0)" +msgstr "**FlagRelease** (v0.1.0)" -#: ../../flagos_homepage/overview.md:126 +#: ../../overview.md:123 msgid "" "FlagRelease is a platform dedicated to the automatic migration, " "adaptation and release of large models for multi-architecture AI chips. " @@ -425,11 +415,11 @@ msgstr "" "FlagRelease 是一个致力于多架构 AI 芯片大模型自动迁移、适配和发布的平台。FlagRelease " "旨在通过自动化、标准化和智能化的适配工作流程,使主流大模型能够以更低成本、更高效率在多样化国产 AI 硬件上完成迁移、验证和发布。" -#: ../../flagos_homepage/overview.md:128 -msgid "**FlagPerf** (v1.2)" -msgstr "**FlagPerf** (v1.2)" +#: ../../overview.md:125 +msgid "**FlagPerf** (v1.2.0)" +msgstr "**FlagPerf** (v1.2.0)" -#: ../../flagos_homepage/overview.md:130 +#: ../../overview.md:127 msgid "" "FlagPerf is an integrated AI hardware evaluation engine. FlagPerf aims to" " establish an industry practice-oriented indicator system and evaluate " @@ -439,33 +429,23 @@ msgstr "" "FlagPerf 是一个集成的 AI 硬件评测引擎。FlagPerf 旨在建立业界实践导向的指标体系,评估 AI 硬件在软件栈组合(模型 + 框架" " + 编译器)下的实际性能。" -#: ../../flagos_homepage/overview.md:132 +#: ../../overview.md:129 msgid "**FlagCICD** (v0.1.0)" msgstr "**FlagCICD** (v0.1.0)" -#: ../../flagos_homepage/overview.md:134 +#: ../../overview.md:131 msgid "" "FlagCICD is a CI/CD toolchain that streamlines large-model development " "across diverse AI chips, eliminating fragmentation and cutting adaptation" " costs." msgstr "FlagCICD 是一个 CI/CD 工具链,简化跨多种 AI 芯片的大模型开发,消除碎片化并降低适配成本。" -#: ../../flagos_homepage/overview.md:136 +#: ../../overview.md:133 msgid "**KernelGenBench** (v0.1.0)" msgstr "**KernelGenBench** (v0.1.0)" -#: ../../flagos_homepage/overview.md:138 +#: ../../overview.md:135 msgid "" "KernelGenBench is a benchmark framework for evaluating LLM and agent-" "based Triton kernel generation across multiple hardware platforms." msgstr "KernelGenBench 是一个基准测试框架,用于评估跨多硬件平台的 LLM 和 Agent 驱动的 Triton kernel 生成能力。" - -#~ msgid "**FlagCX**" -#~ msgstr "**FlagCX**" - -#~ msgid "**FlagCICD**" -#~ msgstr "**FlagCICD**" - -#~ msgid "Ecosystem enablement plugins" -#~ msgstr "生态使能插件" - diff --git a/docs/flagrelease_en/model_list.txt b/docs/flagrelease_en/model_list.txt index 40af92af11..97c060f244 100644 --- a/docs/flagrelease_en/model_list.txt +++ b/docs/flagrelease_en/model_list.txt @@ -205,8 +205,10 @@ FlagRelease/Qwen-Image-2.1-BF16-enflame-FlagOS FlagRelease/Qwen-Image-2.1-BF16-hygon-FlagOS FlagRelease/Qwen-Image-2.1-BF16-metax-FlagOS FlagRelease/Qwen-Image-2.1-BF16-mthreads-FlagOS +FlagRelease/Qwen-Image-2.1-BF16-mthreads-FlagOS-Express FlagRelease/Qwen-Image-2.1-BF16-nvidia-FlagOS FlagRelease/Qwen-Image-2.1-BF16-zhenwu-FlagOS +FlagRelease/Qwen-Image-2.1-BF16-zhenwu-FlagOS-Express FlagRelease/Qwen-Image-2.1-W8A8-arm-FlagOS FlagRelease/Qwen2-7B-FlagOS-Arm FlagRelease/Qwen2-7B-Instruct-FlagOS diff --git a/docs/flagrelease_en/model_list/model-list-aihuanxin.md b/docs/flagrelease_en/model_list/model-list-aihuanxin.md index 35cb327708..91de1f0721 100644 --- a/docs/flagrelease_en/model_list/model-list-aihuanxin.md +++ b/docs/flagrelease_en/model_list/model-list-aihuanxin.md @@ -24,12 +24,28 @@ | GLM-5.2-metax-FlagOS | | | GLM-5.2-mthreads-FlagOS | | | GLM-5.2-zhenwu-FlagOS | | +| GLM-5.3-Flash-BF16-ascend-FlagOS | | +| GLM-5.3-Flash-BF16-tsingmicro-FlagOS | | +| GLM-5.3-Flash-INT8-sunrise-FlagOS | | | gpt-oss-120b-FlagOS | | | grok-2-FlagOS | | | Hunyuan-A13B-Instruct-FlagOS | | +| HY-MT2-1.8B-ascend-FlagOS | | +| HY-MT2-1.8B-nvidia-FlagOS | | +| HY-MT2-1.8B-zhenwu-FlagOS | | +| HY-MT2-30B-A3B-ascend-FlagOS | | +| HY-MT2-30B-A3B-nvidia-FlagOS | | +| HY-MT2-30B-A3B-zhenwu-FlagOS | | +| HY-MT2-7B-ascend-FlagOS | | +| HY-MT2-7B-nvidia-FlagOS | | +| HY-MT2-7B-zhenwu-FlagOS | | | Hy3-hygon-FlagOS | | | Hy3-iluvatar-FlagOS | | +| Hy3-metax-FlagOS | | +| Hy3-mthreads-FlagOS | | +| Hy3-nvidia-FlagOS-Express | | | Hy3-tsingmicro-FlagOS | | +| Hy3-zhenwu-FlagOS | | | Kimi-K2-Thinking-FlagOS | | | MiniCPM-o-4.5-ascend-FlagOS | | | MiniCPM-o-4.5-hygon-FlagOS | | @@ -48,6 +64,18 @@ | MiniCPM5-1B-mthreads-FlagOS | | | MiniCPM5-1B-nvidia-FlagOS | | | MiniCPM5-1B-zhenwu-FlagOS | | +| MiniCPM5-2B-BF16-ascend-FlagOS | | +| MiniCPM5-2B-BF16-hygon-FlagOS | | +| MiniCPM5-2B-BF16-iluvatar-FlagOS | | +| MiniCPM5-2B-BF16-kunlunxin-FlagOS | | +| MiniCPM5-2B-BF16-metax-FlagOS | | +| MiniCPM5-2B-BF16-mthreads-FlagOS | | +| MiniCPM5-2B-BF16-nvidia-FlagOS | | +| MiniCPM5-2B-BF16-sunrise-FlagOS | | +| MiniCPM5-2B-BF16-tsingmicro-FlagOS | | +| MiniCPM5-2B-BF16-zhenwu-FlagOS | | +| MiniCPM5-2B-W4A8-arm-FlagOS | | +| MiniCPM5-2B-W8A8-arm-FlagOS | | | MiniCPM_o_2.6-FlagOS-Cambricon | | | MiniCPM_o_2.6-FlagOS-NVIDIA | | | MiniMax-M2-FlagOS | | @@ -62,6 +90,19 @@ | phi-4-FlagOS | | | phi-4-hygon-FlagOS | | | phi-4-metax-FlagOS | | +| pi0-FlagOS | | +| Qwen-Image-2.1-BF16-ascend-FlagOS | | +| Qwen-Image-2.1-BF16-enflame-FlagOS | | +| Qwen-Image-2.1-BF16-hygon-FlagOS | | +| Qwen-Image-2.1-BF16-metax-FlagOS | | +| Qwen-Image-2.1-BF16-nvidia-FlagOS | | +| Qwen-Image-2.1-BF16-zhenwu-FlagOS | | +| Qwen-Image-2.1-W8A8-arm-FlagOS | | +| Qwen2-7B-FlagOS-Arm | | +| Qwen2-7B-Instruct-FlagOS | | +| Qwen2.5-32B-Instruct-FlagOS-Nvidia | | +| Qwen2.5-VL-32B-Instruct-FlagOS-Metax-BF16 | | +| Qwen2.5-VL-32B-Instruct-FlagOS-Nvidia | | | Qwen3-235B-A22B-FlagOS-nvidia | | | Qwen3-235B-A22B-Instruct-2507-FlagOS | | | Qwen3-30B-A3B-FlagOS-nvidia | | @@ -101,7 +142,34 @@ | Qwen3.8-2.4T-A95B-INT8-nvidia-FlagOS | | | Qwen3.8-2.4T-A95B-INT8-tsingmicro-FlagOS | | | Qwen3.8-2.4T-A95B-INT8-zhenwu-FlagOS | | +| Qwen3.8-Flash-Next-BF16-ascend-FlagOS | | +| Qwen3.8-Flash-Next-BF16-hygon-FlagOS | | +| Qwen3.8-Flash-Next-BF16-iluvatar-FlagOS | | +| Qwen3.8-Flash-Next-BF16-kunlunxin-FlagOS | | +| Qwen3.8-Flash-Next-BF16-metax-FlagOS | | | Qwen3.8-Flash-Next-BF16-mthreads-FlagOS | | +| Qwen3.8-Flash-Next-BF16-nvidia-FlagOS | | +| Qwen3.8-Flash-Next-BF16-tsingmicro-FlagOS | | +| Qwen3.8-Flash-Next-BF16-zhenwu-FlagOS | | | QwQ-32B-FlagOS-Cambricon | | | QwQ-32B-FlagOS-Iluvatar | | | QwQ-32B-FlagOS-Nvidia | | +| RoboBrain-X0-Preview-ascend-FlagOS | | +| RoboBrain-X0-Preview-FlagOS | | +| RoboBrain2.0-32B-Ascend-FlagOS | | +| RoboBrain2.0-32B-FlagOS | | +| RoboBrain2.0-7B-FlagOS | | +| RoboBrain2.0-7B-FlagOS-Ascend | | +| RoboBrain2.0-7B-metax-FlagOS | | +| RoboBrain2.0-7B-W8A16-FlagOS | | +| Seed-OSS-36B-Instruct-FlagOS | | +| step3-FlagOS | | +| Xing4.0-29B-A4B-BF16-ascend-FlagOS | | +| Xing4.0-29B-A4B-BF16-hygon-FlagOS | | +| Xing4.0-29B-A4B-BF16-iluvatar-FlagOS | | +| Xing4.0-29B-A4B-BF16-metax-FlagOS | | +| Xing4.0-29B-A4B-BF16-mthreads-FlagOS | | +| Xing4.0-29B-A4B-BF16-nvidia-FlagOS | | +| Xing4.0-29B-A4B-BF16-tsingmicro-FlagOS | | +| Xing4.0-29B-A4B-BF16-zhenwu-FlagOS | | +| Xing4.0-29B-A4B-W4A8-arm-FlagOS | | diff --git a/docs/flagrelease_en/model_list/model-list-huggingface.md b/docs/flagrelease_en/model_list/model-list-huggingface.md index b34c00e4e8..81034ed3d2 100644 --- a/docs/flagrelease_en/model_list/model-list-huggingface.md +++ b/docs/flagrelease_en/model_list/model-list-huggingface.md @@ -2,18 +2,16 @@ | Model Name | Website | |------------|---------| -| AI21-Jamba-1.5-Mini-metax-FlagOS | | | AI21-Jamba-1.5-Mini-nvidia-FlagOS | | +| BAAI-Cardiac-Agent-hygon-FlagOS | | | DeepSeek-R1-FlagOS-Iluvatar-INT8 | | -| DeepSeek-V3.2-Exp-FlagOS | | +| DeepSeek-V4-Pro-nvidia-FlagOS | | | ERNIE-4.5-0.3B-PT | | | ERNIE-4.5-0.3B-PT-hygon-FlagOS | | +| ERNIE-4.5-0.3B-PT-iluvatar-FlagOS | | +| ERNIE-4.5-0.3B-PT-metax-FlagOS | | | ERNIE-4.5-0.3B-PT-mthreads-FlagOS | | | farm_molecular_representation-hygon-FlagOS | | -| Fathom-R1-14B-mthreads-FlagOS | | -| GLM-5-FP8-FlagOS | | -| GLM-5.2-hygon-FlagOS | | -| GLM-5.2-metax-FlagOS | | | GLM-5.2-mthreads-FlagOS | | | GLM-5.2-zhenwu-FlagOS | | | GLM-5.3-Flash-BF16-ascend-FlagOS | | @@ -24,12 +22,10 @@ | GLM-5.3-Flash-BF16-tsingmicro-FlagOS | | | GLM-5.3-Flash-BF16-zhenwu-FlagOS | | | GLM-5.3-Flash-FP8-mthreads-FlagOS | | -| HY-MT2-1.8B-nvidia-FlagOS | | -| HY-MT2-30B-A3B-mthreads-FlagOS | | +| GLM-5.3-Flash-INT8-sunrise-FlagOS | | | Hy3-iluvatar-FlagOS | | -| Hy3-mthreads-FlagOS | | | Hy3-nvidia-FlagOS-Express | | -| Hy3-zhenwu-FlagOS | | +| Hy3-tsingmicro-FlagOS | | | Hy4-preview-FP8-mthreads-FlagOS | | | Hy4-preview-FP8-nvidia-FlagOS | | | Hy4-preview-INT8-ascend-FlagOS | | @@ -37,31 +33,34 @@ | Hy4-preview-INT8-kunlunxin-FlagOS | | | Hy4-preview-INT8-metax-FlagOS | | | Hy4-preview-INT8-zhenwu-FlagOS | | -| Kimi-Linear-48B-A3B-Instruct-ascend-FlagOS | | | Kimi-Linear-48B-A3B-Instruct-hygon-FlagOS | | | Kimi-Linear-48B-A3B-Instruct-iluvatar-FlagOS | | -| Kimi-Linear-48B-A3B-Instruct-metax-FlagOS | | | Kimi-Linear-48B-A3B-Instruct-nvidia-FlagOS | | +| MiniCPM-o-4.5-ascend-FlagOS | | | MiniCPM-o-4.5-hygon-FlagOS | | +| MiniCPM-o-4.5-iluvatar-FlagOS | | +| MiniCPM-o-4.5-metax-FlagOS | | | MiniCPM-o-4.5-nvidia-FlagOS | | -| MiniCPM5-1B-zhenwu-FlagOS | | -| MiniMax-M2.7-zhenwu-FlagOS | | +| MiniCPM-o-4.5-zhenwu-FlagOS | | +| MiniCPM5-2B-W8A8-arm-FlagOS | | +| MiniMax-M2.7-iluvatar-FlagOS | | +| MiniMax-M2.7-metax-FlagOS | | | MiniMax-M3-ascend-FlagOS | | | MiniMax-M3-hygon-FlagOS | | -| MiniMax-M3-metax-FlagOS | | | MiniMax-M3-mthreads-FlagOS | | -| MiniMax-M3-nvidia-FlagOS | | -| Moonlight-16B-A3B-ascend-FlagOS | | | Moonlight-16B-A3B-hygon-FlagOS | | -| Moonlight-16B-A3B-iluvatar-FlagOS | | | Moonlight-16B-A3B-metax-FlagOS | | | Moonlight-16B-A3B-mthreads-FlagOS | | -| Phi-3.5-MoE-instruct-ascend-FlagOS | | +| Phi-3.5-MoE-instruct-mthreads-FlagOS | | | Phi-3.5-MoE-instruct-nvidia-FlagOS | | -| Qwen3-Coder-Next-nvidia-FlagOS | | -| Qwen3-Next-80B-A3B-Instruct-FlagOS | | +| Qwen-Image-2.1-BF16-ascend-FlagOS | | +| Qwen-Image-2.1-BF16-enflame-FlagOS | | +| Qwen-Image-2.1-BF16-hygon-FlagOS | | +| Qwen-Image-2.1-BF16-metax-FlagOS | | +| Qwen-Image-2.1-BF16-mthreads-FlagOS | | +| Qwen-Image-2.1-BF16-nvidia-FlagOS | | +| Qwen-Image-2.1-BF16-zhenwu-FlagOS | | | Qwen3-Omni-30B-A3B-Instruct-FlagOS | | -| Qwen3.5-397B-A17B-metax-FlagOS | | | Qwen3.5-397B-A17B-nvidia-FlagOS | | | Qwen3.6-27B-metax-FlagOS-Express | | | Qwen3.6-35B-A3B-nomtp-ascend-FlagOS | | @@ -70,15 +69,7 @@ | Qwen3.6-35B-A3B-nomtp-metax-FlagOS-Express | | | Qwen3.6-35B-A3B-nomtp-nvidia-FlagOS | | | Qwen3.8-2.4T-A95B-FP8-enflame-FlagOS | | -| Qwen3.8-2.4T-A95B-FP8-mthreads-FlagOS | | -| Qwen3.8-2.4T-A95B-FP8-nvidia-FlagOS | | -| Qwen3.8-2.4T-A95B-INT8-ascend-FlagOS | | -| Qwen3.8-2.4T-A95B-INT8-hygon-FlagOS | | -| Qwen3.8-2.4T-A95B-INT8-kunlunxin-FlagOS | | | Qwen3.8-2.4T-A95B-INT8-metax-FlagOS | | -| Qwen3.8-2.4T-A95B-INT8-nvidia-FlagOS | | -| Qwen3.8-2.4T-A95B-INT8-tsingmicro-FlagOS | | -| Qwen3.8-2.4T-A95B-INT8-zhenwu-FlagOS | | | Qwen3.8-27B-BF16-ascend-FlagOS | | | Qwen3.8-27B-BF16-enflame-FlagOS | | | Qwen3.8-27B-BF16-hygon-FlagOS-Express | | @@ -102,3 +93,12 @@ | Qwen3.8-Flash-Next-BF16-zhenwu-FlagOS | | | Qwen3.8-Flash-Next-FP8-mthreads-FlagOS | | | RoboBrain-X0-FlagOS | | +| Xing4.0-29B-A4B-BF16-ascend-FlagOS | | +| Xing4.0-29B-A4B-BF16-hygon-FlagOS | | +| Xing4.0-29B-A4B-BF16-iluvatar-FlagOS | | +| Xing4.0-29B-A4B-BF16-metax-FlagOS | | +| Xing4.0-29B-A4B-BF16-mthreads-FlagOS | | +| Xing4.0-29B-A4B-BF16-nvidia-FlagOS | | +| Xing4.0-29B-A4B-BF16-tsingmicro-FlagOS | | +| Xing4.0-29B-A4B-BF16-zhenwu-FlagOS | | +| Xing4.0-29B-A4B-W4A8-arm-FlagOS | | diff --git a/docs/flagrelease_en/model_list/model-list-modelscope.md b/docs/flagrelease_en/model_list/model-list-modelscope.md index 7faf89d915..cae6489ae6 100644 --- a/docs/flagrelease_en/model_list/model-list-modelscope.md +++ b/docs/flagrelease_en/model_list/model-list-modelscope.md @@ -217,8 +217,10 @@ | Qwen-Image-2.1-BF16-hygon-FlagOS | | | Qwen-Image-2.1-BF16-metax-FlagOS | | | Qwen-Image-2.1-BF16-mthreads-FlagOS | | +| Qwen-Image-2.1-BF16-mthreads-FlagOS-Express | | | Qwen-Image-2.1-BF16-nvidia-FlagOS | | | Qwen-Image-2.1-BF16-zhenwu-FlagOS | | +| Qwen-Image-2.1-BF16-zhenwu-FlagOS-Express | | | Qwen-Image-2.1-W8A8-arm-FlagOS | | | Qwen2-7B-FlagOS-Arm | | | Qwen2-7B-Instruct-FlagOS | | diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_DASD-4B-Thinking-hygon-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_DASD-4B-Thinking-hygon-FlagOS.md index 768c3c454a..0ea8ae041e 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_DASD-4B-Thinking-hygon-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_DASD-4B-Thinking-hygon-FlagOS.md @@ -3,6 +3,7 @@ base_model: - "" --- # Introduction +DASD-4B-Thinking is a 4B dense reasoning model from Alibaba-Apsara, post-trained from Qwen3-4B-Instruct-2507 and distilled from gpt-oss-120b for long chain-of-thought reasoning, with 36 layers and a hidden size of 2560. This release packages it with the **FlagOS** software stack for Hygon DCU BW1000 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -15,9 +16,7 @@ base_model: ## Benchmark Result | Metrics | DASD-4B-Thinking-hygon-Origin | DASD-4B-Thinking-hygon-FlagOS | |--------------|-------------------------------|-------------------------------| -| GPQA_Diamond | 44.0 | 60.0 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 44 | 54 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_DASD-4B-Thinking-iluvatar-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_DASD-4B-Thinking-iluvatar-FlagOS.md index 8f1862fddd..4e0d755a9e 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_DASD-4B-Thinking-iluvatar-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_DASD-4B-Thinking-iluvatar-FlagOS.md @@ -3,6 +3,7 @@ base_model: - "" --- # Introduction +DASD-4B-Thinking is a 4B dense reasoning model from Alibaba-Apsara, post-trained from Qwen3-4B-Instruct-2507 and distilled from gpt-oss-120b for long chain-of-thought reasoning, with 36 layers and a hidden size of 2560. This release packages it with the **FlagOS** software stack for Iluvatar BI-V200 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -15,9 +16,7 @@ base_model: ## Benchmark Result | Metrics | DASD-4B-Thinking-iluvatar-Origin | DASD-4B-Thinking-iluvatar-FlagOS | |--------------|----------------------------------|----------------------------------| -| GPQA_Diamond | 44.0 | 66.67 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 44 | 66.67 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_DASD-4B-Thinking-metax-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_DASD-4B-Thinking-metax-FlagOS.md index 2d2a6edcec..43363f4f4c 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_DASD-4B-Thinking-metax-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_DASD-4B-Thinking-metax-FlagOS.md @@ -1,4 +1,5 @@ # Introduction +DASD-4B-Thinking is a 4B dense reasoning model from Alibaba-Apsara, post-trained from Qwen3-4B-Instruct-2507 and distilled from gpt-oss-120b for long chain-of-thought reasoning, with 36 layers and a hidden size of 2560. This release packages it with the **FlagOS** software stack for Metax MetaX C550 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -11,9 +12,7 @@ ## Benchmark Result | Metrics | DASD-4B-Thinking-metax-Origin | DASD-4B-Thinking-metax-FlagOS | |--------------|-------------------------------|-------------------------------| -| GPQA_Diamond | 44.0 | 50.0 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 44 | 60.42 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_GLM-4-32B-0414-hygon-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_GLM-4-32B-0414-hygon-FlagOS.md index 446c83aee6..7dcba7ab09 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_GLM-4-32B-0414-hygon-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_GLM-4-32B-0414-hygon-FlagOS.md @@ -1,5 +1,5 @@ # Introduction - +GLM-4-32B-0414 is a 32B language model from the GLM family by zai-org, built on the Glm4 architecture with 61 layers and a hidden size of 6144, and tuned for instruction following and code. This release packages it with the **FlagOS** software stack for Hygon DCU BW1000 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -10,11 +10,9 @@ # Evaluation Results ## Benchmark Result -| Metrics | GLM-4-32B-0414-hygon-FlagOS-Origin | GLM-4-32B-0414-hygon-FlagOS-FlagOS | -|--------------|------------------------------------|------------------------------------| -| GPQA_Diamond | 0 | 56.57 | -| ERQA | - | - | -| Aime24 | - | - | +| Metrics | GLM-4-32B-0414-hygon-Origin | GLM-4-32B-0414-hygon-FlagOS | +|--------------|-----------------------------|-----------------------------| +| GPQA_Diamond | 55 | 53.54 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_Jan-v1-4B-hygon-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_Jan-v1-4B-hygon-FlagOS.md index 7e535765d6..f5b4433eb2 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_Jan-v1-4B-hygon-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_Jan-v1-4B-hygon-FlagOS.md @@ -3,6 +3,7 @@ base_model: - "" --- # Introduction +Jan-v1-4B is a 4B reasoning and tool-use model from janhq, built on the Qwen3 architecture with 36 layers and a hidden size of 2560, and optimized for the Jan App. This release packages it with the **FlagOS** software stack for Hygon DCU BW1000 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -15,9 +16,7 @@ base_model: ## Benchmark Result | Metrics | Jan-v1-4B-hygon-Origin | Jan-v1-4B-hygon-FlagOS | |--------------|------------------------|------------------------| -| GPQA_Diamond | 64.0 | 68.0 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 64 | 68.0 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_Jan-v1-4B-iluvatar-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_Jan-v1-4B-iluvatar-FlagOS.md index b558eea0e6..4ea5ed6e3b 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_Jan-v1-4B-iluvatar-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_Jan-v1-4B-iluvatar-FlagOS.md @@ -3,6 +3,7 @@ base_model: - "" --- # Introduction +Jan-v1-4B is a 4B reasoning and tool-use model from janhq, built on the Qwen3 architecture with 36 layers and a hidden size of 2560, and optimized for the Jan App. This release packages it with the **FlagOS** software stack for Iluvatar BI-V200 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -15,9 +16,7 @@ base_model: ## Benchmark Result | Metrics | Jan-v1-4B-iluvatar-Origin | Jan-v1-4B-iluvatar-FlagOS | |--------------|---------------------------|---------------------------| -| GPQA_Diamond | 64.0 | 67.35 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 64 | 67.35 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_Jan-v1-4B-metax-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_Jan-v1-4B-metax-FlagOS.md index 521099831e..f8da88399d 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_Jan-v1-4B-metax-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_Jan-v1-4B-metax-FlagOS.md @@ -1,4 +1,5 @@ # Introduction +Jan-v1-4B is a 4B reasoning and tool-use model from janhq, built on the Qwen3 architecture with 36 layers and a hidden size of 2560, and optimized for the Jan App. This release packages it with the **FlagOS** software stack for Metax MetaX C550 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -11,9 +12,7 @@ ## Benchmark Result | Metrics | Jan-v1-4B-metax-Origin | Jan-v1-4B-metax-FlagOS | |--------------|------------------------|------------------------| -| GPQA_Diamond | 64.0 | 68.0 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 64 | 62.0 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_Llama-3.1-Nemotron-70B-Instruct-HF-Metax-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_Llama-3.1-Nemotron-70B-Instruct-HF-Metax-FlagOS.md index b05ab48d52..becd83c38f 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_Llama-3.1-Nemotron-70B-Instruct-HF-Metax-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_Llama-3.1-Nemotron-70B-Instruct-HF-Metax-FlagOS.md @@ -12,11 +12,9 @@ Llama-3.1-Nemotron-70B-Instruct-HF is a large language model from NVIDIA, derive # Evaluation Results ## Benchmark Result -| Metrics | Llama-3.1-Nemotron-70B-Instruct-HF-Metax-FlagOS-Origin | Llama-3.1-Nemotron-70B-Instruct-HF-Metax-FlagOS-FlagOS | -|--------------|----------------------------------------------------------|----------------------------------------------------------| -| GPQA_Diamond | 52.0 | 44.0 | -| ERQA | - | - | -| Aime24 | - | - | +| Metrics | Llama-3.1-Nemotron-70B-Instruct-HF-Metax-Origin | Llama-3.1-Nemotron-70B-Instruct-HF-Metax-FlagOS | +|--------------|-------------------------------------------------|-------------------------------------------------| +| GPQA_Diamond | 52 | 50 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_Llama-3.3-Nemotron-Super-49B-v1.5-Metax-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_Llama-3.3-Nemotron-Super-49B-v1.5-Metax-FlagOS.md index 805de2cae3..298ecbcc0a 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_Llama-3.3-Nemotron-Super-49B-v1.5-Metax-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_Llama-3.3-Nemotron-Super-49B-v1.5-Metax-FlagOS.md @@ -10,11 +10,11 @@ Llama-3.3-Nemotron-Super-49B-v1.5 is a large language model from NVIDIA, derived # Evaluation Results ## Benchmark Result -| Metrics | Llama-3.3-Nemotron-Super-49B-v1.5-Metax-FlagOS-Origin | Llama-3.3-Nemotron-Super-49B-v1.5-Metax-FlagOS-FlagOS | -|--------------|---------------------------------------------------------|---------------------------------------------------------| -| GPQA_Diamond | 76.0 | 76.0 | -| ERQA | - | - | -| Aime24 | - | - | +| Metrics | Llama-3.3-Nemotron-Super-49B-v1.5-Metax-Origin | Llama-3.3-Nemotron-Super-49B-v1.5-Metax-FlagOS | +|--------------|------------------------------------------------|------------------------------------------------| +| GPQA_Diamond | 76.0 | 76.0 | +| ERQA | - | - | +| Aime24 | - | - | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_LocoOperator-4B-hygon-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_LocoOperator-4B-hygon-FlagOS.md index d2b4f5aa7e..9a9e7eca2c 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_LocoOperator-4B-hygon-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_LocoOperator-4B-hygon-FlagOS.md @@ -3,6 +3,7 @@ base_model: - "" --- # Introduction +LocoOperator-4B is a 4B agentic tool-use model from LocoreMind built on the Qwen3 architecture, with 36 layers, a hidden size of 2560, and a 256k-token context window. This release packages it with the **FlagOS** software stack for Hygon DCU BW1000 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -15,9 +16,7 @@ base_model: ## Benchmark Result | Metrics | LocoOperator-4B-hygon-Origin | LocoOperator-4B-hygon-FlagOS | |--------------|------------------------------|------------------------------| -| GPQA_Diamond | 52.0 | 58.0 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 52 | 54.0 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_LocoOperator-4B-iluvatar-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_LocoOperator-4B-iluvatar-FlagOS.md index fb7a6b2e72..1d16103f4d 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_LocoOperator-4B-iluvatar-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_LocoOperator-4B-iluvatar-FlagOS.md @@ -3,6 +3,7 @@ base_model: - "" --- # Introduction +Locooperator-4B is a 4B agentic tool-use model from LocoreMind built on the Qwen3 architecture, with 36 layers, a hidden size of 2560, and a 256k-token context window. This release packages it with the **FlagOS** software stack for Iluvatar BI-V200 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -13,11 +14,9 @@ base_model: # Evaluation Results ## Benchmark Result -| Metrics | LocoOperator-4B-iluvatar-Origin | LocoOperator-4B-iluvatar-FlagOS | +| Metrics | Locooperator-4B-iluvatar-Origin | Locooperator-4B-iluvatar-FlagOS | |--------------|---------------------------------|---------------------------------| -| GPQA_Diamond | 52.0 | 56.0 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 52 | 50 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_LocoOperator-4B-metax-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_LocoOperator-4B-metax-FlagOS.md index 345c7a22c7..c139360896 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_LocoOperator-4B-metax-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_LocoOperator-4B-metax-FlagOS.md @@ -1,4 +1,5 @@ # Introduction +LocoOperator-4B is a 4B agentic tool-use model from LocoreMind built on the Qwen3 architecture, with 36 layers, a hidden size of 2560, and a 256k-token context window. This release packages it with the **FlagOS** software stack for Metax MetaX C550 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -11,9 +12,7 @@ ## Benchmark Result | Metrics | LocoOperator-4B-metax-Origin | LocoOperator-4B-metax-FlagOS | |--------------|------------------------------|------------------------------| -| GPQA_Diamond | 52.0 | 58.0 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 52 | 50 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_Phi-3-vision-128k-instruct-hygon-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_Phi-3-vision-128k-instruct-hygon-FlagOS.md index aaac744f48..ba1df6217a 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_Phi-3-vision-128k-instruct-hygon-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_Phi-3-vision-128k-instruct-hygon-FlagOS.md @@ -3,6 +3,7 @@ base_model: - "" --- # Introduction +Phi-3-vision-128k-instruct is a lightweight multimodal instruction-tuned model from Microsoft in the Phi-3 family that handles text and vision, built on the Phi3VForCausalLM architecture with 32 layers and a hidden size of 3072. This release packages it with the **FlagOS** software stack for Hygon DCU BW1000 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -15,9 +16,7 @@ base_model: ## Benchmark Result | Metrics | Phi-3-vision-128k-instruct-hygon-Origin | Phi-3-vision-128k-instruct-hygon-FlagOS | |--------------|-----------------------------------------|-----------------------------------------| -| GPQA_Diamond | 25.0 | 30.0 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 25 | 30.0 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_Phi-3-vision-128k-instruct-iluvatar-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_Phi-3-vision-128k-instruct-iluvatar-FlagOS.md index e5cb39b8b1..203a6c81f0 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_Phi-3-vision-128k-instruct-iluvatar-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_Phi-3-vision-128k-instruct-iluvatar-FlagOS.md @@ -1,4 +1,5 @@ # Introduction +Phi-3-vision-128k-instruct is a lightweight multimodal instruction-tuned model from Microsoft in the Phi-3 family that handles text and vision, built on the Phi3VForCausalLM architecture with 32 layers and a hidden size of 3072. This release packages it with the **FlagOS** software stack for Iluvatar BI-V200 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -11,9 +12,7 @@ ## Benchmark Result | Metrics | Phi-3-vision-128k-instruct-iluvatar-Origin | Phi-3-vision-128k-instruct-iluvatar-FlagOS | |--------------|--------------------------------------------|--------------------------------------------| -| GPQA_Diamond | 25.0 | 28.0 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 25 | 28.0 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_Phi-3-vision-128k-instruct-metax-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_Phi-3-vision-128k-instruct-metax-FlagOS.md index 5d40e05b86..ae44de2fb0 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_Phi-3-vision-128k-instruct-metax-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_Phi-3-vision-128k-instruct-metax-FlagOS.md @@ -1,4 +1,5 @@ # Introduction +Phi-3-vision-128k-instruct is a lightweight multimodal instruction-tuned model from Microsoft in the Phi-3 family that handles text and vision, built on the Phi3VForCausalLM architecture with 32 layers and a hidden size of 3072. This release packages it with the **FlagOS** software stack for Metax MetaX C550 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -11,9 +12,7 @@ ## Benchmark Result | Metrics | Phi-3-vision-128k-instruct-metax-Origin | Phi-3-vision-128k-instruct-metax-FlagOS | |--------------|-----------------------------------------|-----------------------------------------| -| GPQA_Diamond | 25.0 | 32.0 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 25 | 30.0 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_Qwen-Image-2.1-BF16-mthreads-FlagOS-Express.md b/docs/flagrelease_en/model_readmes/FlagRelease_Qwen-Image-2.1-BF16-mthreads-FlagOS-Express.md new file mode 100644 index 0000000000..1e1236a673 --- /dev/null +++ b/docs/flagrelease_en/model_readmes/FlagRelease_Qwen-Image-2.1-BF16-mthreads-FlagOS-Express.md @@ -0,0 +1,137 @@ +--- +license: apache-2.0 +language: +- zh +- en +--- + +# Introduction +FlagOS is a fully open-source system software stack for heterogeneous AI chips. It unifies the model–system–chip layers to enable a "develop once, run anywhere" workflow, eliminating the fragmentation among vendor-specific software stacks and substantially lowering the cost of porting AI workloads across accelerators. +In this release, Qwen-Image-2.1 leverages the FlagOS software stack to provide direct multi-chip support. By integrating the Triton-based operator library FlagGems via the Torch-FL plugin, FlagOS enables seamless adaptation of the Diffusers library across chip platforms; the usage experience remains identical to that on NVIDIA, requiring zero code modifications. Inference accuracy across all platforms has been aligned with the official implementation. + +### Integrated Deployment +- Out-of-the-box inference scripts with pre-configured hardware and software parameters +- Released **FlagOS-Mthreads** container image supporting deployment within minutes +### Consistency Validation +- Rigorously evaluated through benchmark testing: Performance and results from the FlagOS software stack are compared against native stacks on multiple public. + +# Evaluation Results +## Benchmark Result +| Metrics | Qwen-Image-2.1-Nvidia-Origin | Qwen-Image-2.1-Mthreads-FlagOS | +|--------------|--------------------------------|--------------------------------------| +| T2I-100 (ClipScore) | 33.66 | 33.56 | +| Coco-Image (ClipScore) | 25.34 | 25.38 | + +## Performance Benchmark +| Metric | NV-H100 native-BF16 | Mthreads-BF16 | +| ---- | ---- | ---- | +| TFLOPS (per card) | 989 | 430 | +| Card Count | 1 | 1 | +| TFLOPS (per card) × Card Count | 989 | 430 | +| latency(median), s/image | 6.52 | 14.868851 | +| Throughput (TPS)=1/latency, images/s | 0.153374233 | 0.067254692 | +| text encoder(mean), s | 0.03 | 0.276425 | +| denosing loop(mean), s | 6.32 | 13.48768 | +| vae decoder(mean), s | 0.13 | 1.067318 | +| loop(per step), ms | 158 | 337.192008 | +| torch-fl vs torch-vendor(time) | 1 | 0.126448
=(14.868851/117.588358) | +| peak GiB | 36.82 | 50.781124 GiB | +| Throughput / TFLOPS | 0.00015508 | 0.000156406 (101%)| + +# User Guide +Environment Setup + +| Item | Version | +|------------------|----------------------| +| Docker Version | Docker version 29.3.1 | +| Operating System | 22.04.5 LTS | + +## Operation Steps + +### Download FlagOS Image +```bash +docker pull harbor.baai.ac.cn/flagrelease-public/qwen-image-2.1-mthreads001-gemsnone-treenone-cxnone-pluginnone-vllmnone-sglangnone-sglangflnone-cp310-ptnone-musanone-x64-3.3.5-server:202609240151 +``` + +### Download Open-source Model Weights +```bash +pip install modelscope +modelscope download --model FlagRelease/Qwen-Image-2.1-BF16-mthreads-FlagOS-Express --local_dir /data/Qwen-Image-2.1 +``` + +### Start the Container +```bash +docker run -d \ + --name qwen \ + --runtime=mthreads --network=host --ipc=host --shm-size=32g \ + -e MUSA_VISIBLE_DEVICES=3 \ + -e TORCH_DEVICE_BACKEND_AUTOLOAD=0 \ + -e MODEL_PATH=/public-flash/models/Qwen-Image-2.1 \ + -e OUT_DIR=/output/bench_1024 \ + -v /data/models/Qwen-Image-2.1:/public-flash/models/Qwen-Image-2.1:ro \ + --entrypoint sleep \ + harbor.baai.ac.cn/flagrelease-public/qwen-image-2.1-mthreads001-gemsnone-treenone-cxnone-pluginnone-vllmnone-sglangnone-sglangflnone-cp310-ptnone-musanone-x64-3.3.5-server:202609240151 infinity +``` +### Start the Server +```bash +env -u PYTHONPATH \ + MUSA_VISIBLE_DEVICES=3 \ + TORCH_DEVICE_BACKEND_AUTOLOAD=0 \ + /review/delivery/venv/bin/python \ + /opt/qwen21-under15/qwen21.py generate \ + --prompt '一只橘猫坐在窗边,窗外是雨后的花园,柔和的自然光,真实摄影风格。' \ + --seed 42 \ + --warmup 0 \ + --out /output/custom.png +``` + + + +### AnythingLLM Integration Guide + +#### 1. Download & Install + +- Visit the official site: https://anythingllm.com/ +- Choose the appropriate version for your OS (Windows/macOS/Linux) +- Follow the installation wizard to complete the setup + +#### 2. Configuration + +- Launch AnythingLLM +- Open settings (bottom left, fourth tab) +- Configure core LLM parameters +- Click "Save Settings" to apply changes + +#### 3. Model Interaction + +- After model loading is complete: +- Click **"New Conversation"** +- Enter your question (e.g., "Explain the basics of quantum computing") +- Click the send button to get a response +# Technical Overview +**FlagOS** is a fully open-source system software stack designed to unify the "model–system–chip" layers and foster an open, collaborative ecosystem. It enables a "develop once, run anywhere" workflow across diverse AI accelerators, unlocking hardware performance, eliminating fragmentation among vendor-specific software stacks, and substantially lowering the cost of porting and maintaining AI workloads. With core technologies such as the **FlagScale**, together with vllm-plugin-fl, distributed training/inference framework, **FlagGems** universal operator library, **FlagCX** communication library, and **FlagTree** unified compiler, the **FlagRelease** platform leverages the **FlagOS** stack to automatically produce and release various combinations of . This enables efficient and automated model migration across diverse chips, opening a new chapter for large model deployment and application. +## FlagGems +FlagGems is a high-performance, generic operator libraryimplemented in [Triton](https://github.com/openai/triton) language. It is built on a collection of backend-neutralkernels that aims to accelerate LLM (Large-Language Models) training and inference across diverse hardware platforms. +## FlagTree +FlagTree is an open source, unified compiler for multipleAI chips project dedicated to developing a diverse ecosystem of AI chip compilers and related tooling platforms, thereby fostering and strengthening the upstream and downstream Triton ecosystem. Currently in its initial phase, the project aims to maintain compatibility with existing adaptation solutions while unifying the codebase to rapidly implement single-repository multi-backend support. Forupstream model users, it provides unified compilation capabilities across multiple backends; for downstream chip manufacturers, it offers examples of Triton ecosystem integration. +## FlagScale and vllm-plugin-fl +Flagscale is a comprehensive toolkit designed to supportthe entire lifecycle of large models. It builds on the strengths of several prominent open-source projects, including [Megatron-LM](https://github.com/NVIDIA/Megatron-LM) and [vLLM](https://github.com/vllm-project/vllm), to provide a robust, end-to-end solution for managing and scaling large models. +vllm-plugin-fl is a vLLM plugin built on the FlagOS unified multi-chip backend, to help flagscale support multi-chip on vllm framework. +## **FlagCX** +FlagCX is a scalable and adaptive cross-chip communication library. It serves as a platform where developers, researchers, and AI engineers can collaborate on various projects, contribute to the development of cutting-edge AI solutions, and share their work with the global community. + +## **FlagEval Evaluation Framework** + FlagEval is a comprehensive evaluation system and open platform for large models launched in 2023. It aims to establish scientific, fair, and open benchmarks, methodologies, and tools to help researchers assess model and training algorithm performance. It features: + - **Multi-dimensional Evaluation**: Supports 800+ modelevaluations across NLP, CV, Audio, and Multimodal fields,covering 20+ downstream tasks including language understanding and image-text generation. + - **Industry-Grade Use Cases**: Has completed horizonta1 evaluations of mainstream large models, providing authoritative benchmarks for chip-model performance validation. + +# Contributing + +We warmly welcome global developers to join us: + +1. Submit Issues to report problems +2. Create Pull Requests to contribute code +3. Improve technical documentation +4. Expand hardware adaptation support +# License +The model weights are derived from Qwen/Qwen-Image-2.1 and are open‑sourced under the Apache License 2.0: https://www.apache.org/licenses/LICENSE-2.0.txt diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_Qwen-Image-2.1-BF16-mthreads-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_Qwen-Image-2.1-BF16-mthreads-FlagOS.md index 4c063c7acf..cd2459c2d9 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_Qwen-Image-2.1-BF16-mthreads-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_Qwen-Image-2.1-BF16-mthreads-FlagOS.md @@ -34,7 +34,7 @@ Environment Setup ### Download FlagOS Image ```bash -docker pull harbor.baai.ac.cn/flagrelease-public/qwen-image-2.1-mthreads001-gemsnone-treenone-cxnone-pluginnone-vllmnone-sglangnone-sglangflnone-cp310-ptnone-musanone-x64-3.3.5-server:202609200725 +docker pull harbor.baai.ac.cn/flagrelease-public/qwen-image-2.1-mthreads001-gemsnone-treenone-cxnone-pluginnone-vllmnone-sglangnone-sglangflnone-cp310-ptnone-musanone-x64-3.3.5-server:202609240151 ``` ### Download Open-source Model Weights @@ -54,38 +54,19 @@ docker run -d \ -e OUT_DIR=/output/bench_1024 \ -v /data/models/Qwen-Image-2.1:/public-flash/models/Qwen-Image-2.1:ro \ --entrypoint sleep \ - harbor.baai.ac.cn/flagrelease-public/qwen-image-2.1-mthreads001-gemsnone-treenone-cxnone-pluginnone-vllmnone-sglangnone-sglangflnone-cp310-ptnone-musanone-x64-3.3.5-server:202609200725 infinity + harbor.baai.ac.cn/flagrelease-public/qwen-image-2.1-mthreads001-gemsnone-treenone-cxnone-pluginnone-vllmnone-sglangnone-sglangflnone-cp310-ptnone-musanone-x64-3.3.5-server:202609240151 infinity ``` ### Start the Server ```bash -cd /opt/qwen21-no-fbc-optimized/pr342-source - -export PYTHON=/review/delivery/venv/bin/python -export PYTHONPATH=/review/pr342-autoload:/review/pr342-site:/opt/qwen21-runtime:/review/pr342-diffusers:$PWD -export TORCH_DEVICE_BACKEND_AUTOLOAD=0 -export LD_LIBRARY_PATH=/usr/local/musa/lib:/usr/local/musa/lib64:${LD_LIBRARY_PATH:-} -export QWEN_IMAGE_21_MODEL=/data/Qwen-Image-2.1 -export QWEN_IMAGE_21_DIFFUSERS=/review/pr342-diffusers -export TRITON_CACHE_DIR=/opt/qwen21-triton-cache -export PYTHONWARNINGS=ignore -export FLAGGEMS_LIBENTRY_DEFAULT_BENCHMARK=event -export FLAGOS_ACCELERATOR=musa -export FLAGOS_LOG=fallback -export FLAGOS_OP_randn=musa -export FLAGOS_OP_convolution_overrideable=flaggems -export FLAGOS_OP_add__Tensor=flaggems -export FLAGOS_OP_div__Tensor=flaggems -export FLAGOS_OP_add___Tensor=flaggems -export FLAGOS_OP_sub__Tensor=flaggems -export LOG=/output/bench_1024/bench.log - -mkdir -p /output/bench_1024 -tests/manual/qwen_image_21/run.sh bench \ - --device flagos --batch 1 --steps 40 \ - --height 1024 --width 1024 --seed 42 --true-cfg-scale 1 \ - --warmup 2 --min-run-time 60 \ - --out /output/bench_1024/result.json \ - --image /output/bench_1024/result.png +env -u PYTHONPATH \ + MUSA_VISIBLE_DEVICES=3 \ + TORCH_DEVICE_BACKEND_AUTOLOAD=0 \ + /review/delivery/venv/bin/python \ + /opt/qwen21-under15/qwen21.py generate \ + --prompt '一只橘猫坐在窗边,窗外是雨后的花园,柔和的自然光,真实摄影风格。' \ + --seed 42 \ + --warmup 0 \ + --out /output/custom.png ``` diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_Qwen-Image-2.1-BF16-zhenwu-FlagOS-Express.md b/docs/flagrelease_en/model_readmes/FlagRelease_Qwen-Image-2.1-BF16-zhenwu-FlagOS-Express.md new file mode 100644 index 0000000000..d5f86bad35 --- /dev/null +++ b/docs/flagrelease_en/model_readmes/FlagRelease_Qwen-Image-2.1-BF16-zhenwu-FlagOS-Express.md @@ -0,0 +1,140 @@ +--- +license: apache-2.0 +language: +- zh +- en +--- + +# Introduction +FlagOS is a fully open-source system software stack for heterogeneous AI chips. It unifies the model–system–chip layers to enable a "develop once, run anywhere" workflow, eliminating the fragmentation among vendor-specific software stacks and substantially lowering the cost of porting AI workloads across accelerators. +In this release, Qwen-Image-2.1 leverages the FlagOS software stack to provide direct multi-chip support. By integrating the Triton-based operator library FlagGems via the Torch-FL plugin, FlagOS enables seamless adaptation of the Diffusers library across chip platforms; the usage experience remains identical to that on NVIDIA, requiring zero code modifications. Inference accuracy across all platforms has been aligned with the official implementation. + +### Integrated Deployment +- Out-of-the-box inference scripts with pre-configured hardware and software parameters +- Released **FlagOS-T-Head** container image supporting deployment within minutes +### Consistency Validation +- Rigorously evaluated through benchmark testing: Performance and results from the FlagOS software stack are compared against native stacks on multiple public. + +# Evaluation Results +## Benchmark Result +| Metrics | Qwen-Image-2.1-Nvidia-Origin | Qwen-Image-2.1-T-Head-FlagOS | +|--------------|--------------------------------|--------------------------------------| +| T2I-100 (ClipScore) | 33.66 | 33.80 | +| Coco-Image (ClipScore) | 25.34 | 25.25 | + +## Performance Benchmark +| Metric | NV-H100 native-BF16 | T-head-BF16 | +| ---- | ---- | ---- | +| TFLOPS (per card) | 989 | 123 | +| Card Count | 1 | 1 | +| TFLOPS (per card) × Card Count | 989 | 123 | +| latency(median), s/image | 6.52 | 37.44 | +| Throughput (TPS)=1/latency, images/s | 0.153374233 | 0.026709402 | +| text encoder(mean), s | 0.03 | 0.116 | +| denosing loop(mean), s | 6.32 | 36.86 | +| vae decoder(mean), s | 0.13 | 0.427 | +| loop(per step), ms | 158 | 921.5 | +| torch-fl vs torch-vendor(time) | 1 | 1.033
(37.44/36.24) | +| peak GiB | 36.82 | 36.8 | +| Throughput / TFLOPS | 0.00015508 | 0.00021715 (140%)| + +# User Guide +Environment Setup + +| Item | Version | +|------------------|----------------------| +| Docker Version | Docker version 29.7.2 | +| Operating System | 24.04.2 LTS | + +## Operation Steps + +### Download FlagOS Image +```bash +docker pull harbor.baai.ac.cn/flagrelease-public/qwen-image2.1-pp001-gems0.0.0-tree0.6.2-cxnone-pluginnone-vllm0.23.0-cp312-pt210-hggc130-x64-2.1.1-rbd225:202609201422 + +``` + +### Download Open-source Model Weights +```bash +pip install modelscope +modelscope download --model FlagRelease/Qwen-Image-2.1-BF16-zhenwu-FlagOS-Express --local_dir /data/Qwen-Image-2.1 +``` + +### Start the Container +```bash +docker run --privileged -dit \ + -e HOST_HOSTNAME=$(hostname) \ + --network=host \ + --device=/dev/infiniband \ + --ipc=host \ + --device=/dev/alixpu_ctl \ + --device=/dev/alixpu \ + --ulimit memlock=-1 \ + --ulimit stack=67108864 \ + --init \ + -v /data:/data \ + -v /mnt:/mnt \ + --name qwen-image \ + harbor.baai.ac.cn/flagrelease-public/qwen-image2.1-pp001-gems0.0.0-tree0.6.2-cxnone-pluginnone-vllm0.23.0-cp312-pt210-hggc130-x64-2.1.1-rbd225:202609201422 + +``` +### Start the Server +```bash +bash /workspace/PyTorch-Plugin-FL/tests/manual/qwen_image_21/run.sh infer \ + --device flagos \ + --devices flagos:0 \ + --model /data/Qwen-Image-2.1 \ + --run-dir /workspace \ + --steps 40 \ + --log /workspace/infer-flagos.log +``` + + +### AnythingLLM Integration Guide + +#### 1. Download & Install + +- Visit the official site: https://anythingllm.com/ +- Choose the appropriate version for your OS (Windows/macOS/Linux) +- Follow the installation wizard to complete the setup + +#### 2. Configuration + +- Launch AnythingLLM +- Open settings (bottom left, fourth tab) +- Configure core LLM parameters +- Click "Save Settings" to apply changes + +#### 3. Model Interaction + +- After model loading is complete: +- Click **"New Conversation"** +- Enter your question (e.g., "Explain the basics of quantum computing") +- Click the send button to get a response +# Technical Overview +**FlagOS** is a fully open-source system software stack designed to unify the "model–system–chip" layers and foster an open, collaborative ecosystem. It enables a "develop once, run anywhere" workflow across diverse AI accelerators, unlocking hardware performance, eliminating fragmentation among vendor-specific software stacks, and substantially lowering the cost of porting and maintaining AI workloads. With core technologies such as the **FlagScale**, together with vllm-plugin-fl, distributed training/inference framework, **FlagGems** universal operator library, **FlagCX** communication library, and **FlagTree** unified compiler, the **FlagRelease** platform leverages the **FlagOS** stack to automatically produce and release various combinations of . This enables efficient and automated model migration across diverse chips, opening a new chapter for large model deployment and application. +## FlagGems +FlagGems is a high-performance, generic operator libraryimplemented in [Triton](https://github.com/openai/triton) language. It is built on a collection of backend-neutralkernels that aims to accelerate LLM (Large-Language Models) training and inference across diverse hardware platforms. +## FlagTree +FlagTree is an open source, unified compiler for multipleAI chips project dedicated to developing a diverse ecosystem of AI chip compilers and related tooling platforms, thereby fostering and strengthening the upstream and downstream Triton ecosystem. Currently in its initial phase, the project aims to maintain compatibility with existing adaptation solutions while unifying the codebase to rapidly implement single-repository multi-backend support. Forupstream model users, it provides unified compilation capabilities across multiple backends; for downstream chip manufacturers, it offers examples of Triton ecosystem integration. +## FlagScale and vllm-plugin-fl +Flagscale is a comprehensive toolkit designed to supportthe entire lifecycle of large models. It builds on the strengths of several prominent open-source projects, including [Megatron-LM](https://github.com/NVIDIA/Megatron-LM) and [vLLM](https://github.com/vllm-project/vllm), to provide a robust, end-to-end solution for managing and scaling large models. +vllm-plugin-fl is a vLLM plugin built on the FlagOS unified multi-chip backend, to help flagscale support multi-chip on vllm framework. +## **FlagCX** +FlagCX is a scalable and adaptive cross-chip communication library. It serves as a platform where developers, researchers, and AI engineers can collaborate on various projects, contribute to the development of cutting-edge AI solutions, and share their work with the global community. + +## **FlagEval Evaluation Framework** + FlagEval is a comprehensive evaluation system and open platform for large models launched in 2023. It aims to establish scientific, fair, and open benchmarks, methodologies, and tools to help researchers assess model and training algorithm performance. It features: + - **Multi-dimensional Evaluation**: Supports 800+ modelevaluations across NLP, CV, Audio, and Multimodal fields,covering 20+ downstream tasks including language understanding and image-text generation. + - **Industry-Grade Use Cases**: Has completed horizonta1 evaluations of mainstream large models, providing authoritative benchmarks for chip-model performance validation. + +# Contributing + +We warmly welcome global developers to join us: + +1. Submit Issues to report problems +2. Create Pull Requests to contribute code +3. Improve technical documentation +4. Expand hardware adaptation support +# License +The model weights are derived from Qwen/Qwen-Image-2.1 and are open‑sourced under the Apache License 2.0: https://www.apache.org/licenses/LICENSE-2.0.txt diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_Qwen2.5-Coder-7B-Instruct-ascend-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_Qwen2.5-Coder-7B-Instruct-ascend-FlagOS.md index b3bc7b85f1..61cf0de513 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_Qwen2.5-Coder-7B-Instruct-ascend-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_Qwen2.5-Coder-7B-Instruct-ascend-FlagOS.md @@ -1,5 +1,5 @@ # Introduction - +Qwen2.5-Coder-7B-Instruct is a 7B code-specific instruction-tuned model from the Qwen2.5-Coder series, built on the Qwen2ForCausalLM architecture with 28 layers and a hidden size of 3584. This release packages it with the **FlagOS** software stack for Ascend 910C accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -10,11 +10,9 @@ # Evaluation Results ## Benchmark Result -| Metrics | Qwen2.5-Coder-7B-Instruct-ascend-FlagOS-Origin | Qwen2.5-Coder-7B-Instruct-ascend-FlagOS-FlagOS | -|--------------|------------------------------------------------|------------------------------------------------| -| GPQA_Diamond | 38.0 | 32.0 | -| ERQA | - | - | -| Aime24 | - | - | +| Metrics | Qwen2.5-Coder-7B-Instruct-ascend-Origin | Qwen2.5-Coder-7B-Instruct-ascend-FlagOS | +|--------------|-----------------------------------------|-----------------------------------------| +| GPQA_Diamond | 27 | 34.0 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_granite-4.0-micro-hygon-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_granite-4.0-micro-hygon-FlagOS.md index 382822bb68..b0106b3577 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_granite-4.0-micro-hygon-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_granite-4.0-micro-hygon-FlagOS.md @@ -3,6 +3,7 @@ base_model: - "" --- # Introduction +granite-4.0-micro is a compact instruction-tuned model from IBM Granite built on the GraniteMoeHybrid architecture, which combines dense and Mixture-of-Experts layers, with 40 layers and a hidden size of 2560. This release packages it with the **FlagOS** software stack for Hygon DCU BW1000 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -15,9 +16,7 @@ base_model: ## Benchmark Result | Metrics | granite-4.0-micro-hygon-Origin | granite-4.0-micro-hygon-FlagOS | |--------------|--------------------------------|--------------------------------| -| GPQA_Diamond | 30.0 | 26.0 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 30 | 32.0 | # User Guide Environment Setup diff --git a/docs/flagrelease_en/model_readmes/FlagRelease_granite-4.0-micro-iluvatar-FlagOS.md b/docs/flagrelease_en/model_readmes/FlagRelease_granite-4.0-micro-iluvatar-FlagOS.md index 4f1208269e..38a0fdaf3c 100644 --- a/docs/flagrelease_en/model_readmes/FlagRelease_granite-4.0-micro-iluvatar-FlagOS.md +++ b/docs/flagrelease_en/model_readmes/FlagRelease_granite-4.0-micro-iluvatar-FlagOS.md @@ -5,6 +5,7 @@ frameworks: - "" --- # Introduction +granite-4.0-micro is a compact instruction-tuned model from IBM Granite built on the GraniteMoeHybrid architecture, which combines dense and Mixture-of-Experts layers, with 40 layers and a hidden size of 2560. This release packages it with the **FlagOS** software stack for Iluvatar BI-V200 accelerators, delivering an out-of-the-box, containerized deployment. ### Integrated Deployment - Out-of-the-box inference scripts with pre-configured hardware and software parameters @@ -17,9 +18,7 @@ frameworks: ## Benchmark Result | Metrics | granite-4.0-micro-iluvatar-Origin | granite-4.0-micro-iluvatar-FlagOS | |--------------|-----------------------------------|-----------------------------------| -| GPQA_Diamond | 30.0 | 28.0 | -| ERQA | - | - | -| Aime24 | - | - | +| GPQA_Diamond | 30 | 28.0 | # User Guide Environment Setup diff --git a/docs/flagtree_en/release_notes/release_notes_v070.md b/docs/flagtree_en/release_notes/release_notes_v070.md index 8adb292370..c5d004c466 100644 --- a/docs/flagtree_en/release_notes/release_notes_v070.md +++ b/docs/flagtree_en/release_notes/release_notes_v070.md @@ -1,28 +1,27 @@ -# FlagTree 0.7.0 Release - -- **Added Features** - - 3.6.x branch: - - TLE-Lite: - - Added the `tle.shard_id` op for querying the coordinate of the current program along a device-mesh axis. Supported on NVIDIA. - - Added the following distributed ops: `tle.signal` and `tle.signal_wait`. Supported on NVIDIA. - - Extended `tle.remote` with the `space` (cluster / device / node), `dtype`, `offset`, `coopkind`, and `netidx` parameters, covering remote access within a thread-block cluster (DSMEM), between GPUs in a node (NVLink P2P), and across nodes (FlagCX/RDMA). Supported on NVIDIA. - - Extended `tle.distributed_barrier` with the `space`, `group_kind`, `barrier_kind`, and `order` parameters. Supported on NVIDIA. - - TLE-Struct: - - Added GPU buffer aliasing through `tle.gpu.alloc(..., alias=...)`, providing typed shared-memory views with static validation of aliased views. - - Added the `tle.gpu.set_layout` op for explicit distributed-layout assignment, together with the `BlockEncoding`, `MmaEncoding`, `DotOperandEncoding`, and `SlicedEncoding` layout objects. - - Added the following barrier ops: `tle.gpu.alloc_barrier`, `tle.gpu.alloc_barriers`, `tle.gpu.barrier_wait`, and `tle.gpu.barrier_arrive`; added the `barrier` and `mask` parameters to `tle.gpu.copy`. Supported on NVIDIA. - - Added the `tle.gpu.wgmma` and `tle.gpu.wgmma_wait` ops. Supported on NVIDIA. - - Added the `tle.gpu.buffered_tensor.slot` and `tle.gpu.buffered_tensor.reshape` ops. Supported on NVIDIA. - - Added the `init_value` and `alias_offset_bytes` parameters to `tle.gpu.alloc`. Supported on NVIDIA. - - TLE-Raw: - - Added the `library` and `compiler` parameters, enabling NVSHMEM device-side interfaces to be inlined into TLE-Raw kernels via `@dialect(..., library="nvshmem", compiler="clang")`. Supported on NVIDIA. - - - Backends: - - Added the following backend integrations (based on Triton 3.6) and added CI/CD: [tileir](/getting_started/multi-backend-prebuilt-docker-image-install/install-tileir.md) (NVIDIA TileIR), [ppu](/getting_started/multi-backend-prebuilt-docker-image-install/install-ppu.md) (T-Head), and [spacemit](/getting_started/multi-backend-prebuilt-docker-image-install/install-spacemit.md) (SpacemiT). - - Upgraded the following backends to Triton 3.6 and added CI/CD: [sunrise](/getting_started/multi-backend-prebuilt-docker-image-install/install-sunrise.md), [xpu](/getting_started/multi-backend-prebuilt-docker-image-install/install-xpu.md), [iluvatar](/getting_started/multi-backend-prebuilt-docker-image-install/install-iluvatar.md), and [tsingmicro](/getting_started/multi-backend-prebuilt-docker-image-install/install-tsingmicro.md). - - Added TLE support for the [amd](/getting_started/multi-backend-prebuilt-docker-image-install/install-amd.md) backend and added CI/CD. - - [rpu](/getting_started/multi-backend-prebuilt-docker-image-install/install-rpu.md) (Huixi Intelligence, Triton 3.6) is also supported. On the 3.3.x branch, [ARM64 CPU](/getting_started/install-arm64-cpu.md) provides [TLE-CPU](/user_guide/use-tle-cpu.md). - - [ARM64 CPU](/getting_started/flagtree-cpu.md) provides a CPU backend based on Triton 3.7.2, validated on Linux Arm64 with a vector-add kernel and FlagGems W4A8 operator tests. - -- **DevTools (Debugger & Profiler)** - - FlagPrism ([flagos-ai/FlagPrism](https://github.com/flagos-ai/FlagPrism)) provides debugging and performance-analysis tools for Triton programs, containing `flagtree.debugger` and `flagtree.profiler`, and is integrated into FlagTree as the `third_party/FlagPrism` submodule. It initially supports a subset of backends: Huawei Ascend, Iluvatar, and Moore Threads. +# FlagTree 0.7.0 Release + +- **Added Features** + - 3.6.x branch: + - TLE-Lite: + - Added the `tle.shard_id` op for querying the coordinate of the current program along a device-mesh axis. Supported on NVIDIA. + - Added the following distributed ops: `tle.signal` and `tle.signal_wait`. Supported on NVIDIA. + - Extended `tle.remote` with the `space` (cluster / device / node), `dtype`, `offset`, `coopkind`, and `netidx` parameters, covering remote access within a thread-block cluster (DSMEM), between GPUs in a node (NVLink P2P), and across nodes (FlagCX/RDMA). Supported on NVIDIA. + - Extended `tle.distributed_barrier` with the `space`, `group_kind`, `barrier_kind`, and `order` parameters. Supported on NVIDIA. + - TLE-Struct: + - Added GPU buffer aliasing through `tle.gpu.alloc(..., alias=...)`, providing typed shared-memory views with static validation of aliased views. + - Added the `tle.gpu.set_layout` op for explicit distributed-layout assignment, together with the `BlockEncoding`, `MmaEncoding`, `DotOperandEncoding`, and `SlicedEncoding` layout objects. + - Added the following barrier ops: `tle.gpu.alloc_barrier`, `tle.gpu.alloc_barriers`, `tle.gpu.barrier_wait`, and `tle.gpu.barrier_arrive`; added the `barrier` and `mask` parameters to `tle.gpu.copy`. Supported on NVIDIA. + - Added the `tle.gpu.wgmma` and `tle.gpu.wgmma_wait` ops. Supported on NVIDIA. + - Added the `tle.gpu.buffered_tensor.slot` and `tle.gpu.buffered_tensor.reshape` ops. Supported on NVIDIA. + - Added the `init_value` and `alias_offset_bytes` parameters to `tle.gpu.alloc`. Supported on NVIDIA. + - TLE-Raw: + - Added the `library` and `compiler` parameters, enabling NVSHMEM device-side interfaces to be inlined into TLE-Raw kernels via `@dialect(..., library="nvshmem")`. Supported on NVIDIA. + + - Backends: + - Added the following backend integrations (based on Triton 3.6): [tileir](/getting_started/multi-backend-prebuilt-docker-image-install/install-tileir.md) (NVIDIA TileIR), [ppu](/getting_started/multi-backend-prebuilt-docker-image-install/install-ppu.md) (T-Head), and [spacemit](/getting_started/multi-backend-prebuilt-docker-image-install/install-spacemit.md) (SpacemiT). + - Upgraded the following backends to Triton 3.6: [sunrise](/getting_started/multi-backend-prebuilt-docker-image-install/install-sunrise.md), [xpu](/getting_started/multi-backend-prebuilt-docker-image-install/install-xpu.md), [iluvatar](/getting_started/multi-backend-prebuilt-docker-image-install/install-iluvatar.md), and [tsingmicro](/getting_started/multi-backend-prebuilt-docker-image-install/install-tsingmicro.md). + - Added TLE support for the [amd](/getting_started/multi-backend-prebuilt-docker-image-install/install-amd.md) backend. + - [ARM64 CPU](/getting_started/flagtree-cpu.md) provides a CPU backend based on Triton 3.7.2, validated on Linux Arm64 with a vector-add kernel and FlagGems W4A8 operator tests. + +- **DevTools (Debugger & Profiler)** + - FlagPrism ([flagos-ai/FlagPrism](https://github.com/flagos-ai/FlagPrism)) provides debugging and performance-analysis tools for Triton programs, containing `flagtree.debugger` and `flagtree.profiler`, and is integrated into FlagTree as the `third_party/FlagPrism` submodule. It initially supports a subset of backends: Huawei Ascend, Iluvatar, and Moore Threads. diff --git a/docs/flagtree_en/user_guide/use-tle-raw.md b/docs/flagtree_en/user_guide/use-tle-raw.md index 524985c103..1baf21d4a9 100644 --- a/docs/flagtree_en/user_guide/use-tle-raw.md +++ b/docs/flagtree_en/user_guide/use-tle-raw.md @@ -1,207 +1,207 @@ -# Use TLE-Raw - -This section introduces how to use TLE-Raw. TLE-Raw is available on trition_3.6.x branch. - -TLE Raw provides a low-level extension interface for Triton, allowing users to fill capability gaps and gain fine-grained control via third-party dialects and languages (for example, using CUDA for thread-level scheduling, synchronization, and memory access). Users can choose between portability and composable optimization (via MLIR dialect integration) and maximum fine-grained control (via CUDA integration) based on the target hardware and toolchain maturity. - -## Integrate MLIR dialect into LLVM for portability and composable optimization - -The following is an example of MLIR (Multi-Level Intermediate Representation). - -```{code-block} python -from typing_extensions import Literal as L - -from mlir import ir -from mlir.dialects import arith, llvm, nvvm, scf -import torch -import triton -import triton.language as tl -from triton.experimental.tle.raw import dialect, Input -import triton.experimental.tle.language.raw as tle_raw - -DEVICE = triton.runtime.driver.active.get_active_torch_device() - -@dialect(name="mlir") -def vector_add_tile( - output: Input[L["!llvm.ptr<1>"]], - x: Input[L["!llvm.ptr<1>"]], - y: Input[L["!llvm.ptr<1>"]], - n_elements: Input[L["i32"]], -): - tidx = nvvm.read_ptx_sreg_tid_x(ir.IntegerType.get_signless(32)) - bdimx = nvvm.read_ptx_sreg_ntid_x(ir.IntegerType.get_signless(32)) - gdimx = nvvm.read_ptx_sreg_nctaid_x(ir.IntegerType.get_signless(32)) - bidx = nvvm.read_ptx_sreg_ctaid_x(ir.IntegerType.get_signless(32)) - tidx = arith.index_cast(ir.IndexType.get(), tidx) - bdimx = arith.index_cast(ir.IndexType.get(), bdimx) - gdimx = arith.index_cast(ir.IndexType.get(), gdimx) - bidx = arith.index_cast(ir.IndexType.get(), bidx) - idx = arith.addi(arith.muli(bidx, bdimx), tidx) - step = arith.muli(bdimx, gdimx) - n_elements = arith.index_cast(ir.IndexType.get(), n_elements) - for i in scf.for_(idx, n_elements, step): - i = arith.index_cast(ir.IntegerType.get_signless(32), i) - ptrty = ir.Type.parse("!llvm.ptr<1>") - f32ty = ir.Type.parse("f32") - xptr = llvm.getelementptr(ptrty, x, [i], [-2147483648], f32ty, 0) - yptr = llvm.getelementptr(ptrty, y, [i], [-2147483648], f32ty, 0) - xval = llvm.load(f32ty, xptr) - yval = llvm.load(f32ty, yptr) - outval = arith.addf(xval, yval) - outptr = llvm.getelementptr(ptrty, output, [i], [-2147483648], f32ty, 0) - llvm.store(outval, outptr) - scf.yield_([]) - -@triton.jit -def add_kernel( - x_ptr, - y_ptr, - output_ptr, - n_elements, - BLOCK_SIZE: tl.constexpr, -): - tle_raw.call(vector_add_tile, [output_ptr, x_ptr, y_ptr, n_elements]) - -def add(x: torch.Tensor, y: torch.Tensor): - output = torch.empty_like(x) - assert x.device == DEVICE and y.device == DEVICE and output.device == DEVICE - n_elements = output.numel() - grid = lambda meta: (triton.cdiv(n_elements, meta["BLOCK_SIZE"]), ) - add_kernel[grid](x, y, output, n_elements, BLOCK_SIZE=1024) - return output - -if __name__ == "__main__": - x = torch.randn(2048, device=DEVICE) - y = torch.randn(2048, device=DEVICE) - z = add(x, y) - assert torch.allclose(x + y, z), (x + y, z) -``` - -TLE-raw consists of the following parts: - -- Dialect declaration (decorator) - - Decorator: @tle.raw.language(name="mlir") - - Explanation: This decorator marks the function vector_add_tile as a block of code written directly in the MLIR dialect. It tells the compiler, specifically through the FlagTree EDSL (Embedded Domain Specific Language), that the body of this function should be interpreted and lowered using MLIR operations (such as nvvm, arith, and tensor), rather than standard Python or Triton operations. -- Function implementation - - Function: vector_add_tile(...) - - Explanation: This is the actual implementation of the computation kernel written using low-level MLIR Python bindings. It defines the specific operations (thread indexing, memory loading, floating-point addition, and memory storing) that will be executed by the hardware. -- Function call - - Invocation: tle_raw.call(vector_add_tile, args=[x, y, output]) - - Explanation: This line invokes the declared MLIR function (vector_add_tile) from within the high-level Triton kernel (add_kernel). It passes the input tensors x, y, and the output buffer. Crucially, it provides hardware mapping hints (defining the number of threads) and memory layout specifications (defining the tensors as residing in "shared" memory with a specific order). This allows the compiler to bridge the gap between the high-level tl.load/tl.store operations and the low-level MLIR IR generation. - -## Integrate CUDA into LLVM for maximum fine-grained control - -This section only covers how to integrate CUDA kernel into LLVM inline path. Integrating other vendors into LLVM inline path can follow similar steps. - -TLE-Raw supports CUDA kernel integration via the LLVM inline path. Vendors integrating TLE-Raw on the CUDA side should evaluate: - -- Whether clang can generate LLVM IR and serialize it as text -- Whether TTGIR-related pass operations can be reused or adapted - - -### LLVM route - -Basic flow: use clang to translate CUDA code into LLVM IR, then apply the existing LLVM inline pass. - -![alt text](../assets/images/cuda-to-vllm-line-pass.png) - -### Usage example - -- Reference: `python/tutorials/tle/raw/cuda/01-vector-add.py` - -- Triton side: provide the CUDA file path and function declaration. Other vendors could register your own language name as the value of the `name` parameter. - ![alt text](../assets/images/triton-side.png) - - -- CUDA side: implement the CUDA kernel. LLVM struct parameter declarations are still retained (because subsequent inlining requires handling the Triton ptr-to-LLVM conversion, which is currently left to the user for one-to-one mapping). Other vendors should customize the mapping based on your own language. - - ![alt text](../assets/images/cuda-side.PNG) - - -### Multi-GPU communication with NVSHMEM - -NVSHMEM is NVIDIA's PGAS (Partitioned Global Address Space) communication library for multi-GPU and multi-node programs. It gives every rank a unified address space: a region of memory on each GPU (symmetric memory) is reachable both from the local GPU and directly from other GPUs, so cross-GPU communication can be expressed with ordinary load/store or put/get semantics instead of hand-written point-to-point send/receive logic. It provides both host-side and device-side APIs, and the device-side interfaces can be called from inside a kernel, which makes it a natural fit for all-gather, all-reduce, and fused GEMM-plus-communication patterns. NVSHMEM is used only for NVIDIA GPUs; the distributed primitives of TLE-Lite and TLE-Struct use FlagCX instead of NVSHMEM. - -On top of the CUDA integration, `@dialect` adds the `library` and `compiler` parameters so that NVSHMEM device-side interfaces can be inlined directly into a TLE-Raw kernel: - -```{code-block} python -@dialect( - name="cuda", - library="nvshmem", - compiler="clang", - file=(Path(__file__).parent / "simple-shift-device.cu"), - extern_func_name="simple_shift", -) -def simple_shift(*args, **kwargs): - ... -``` - -- `library="nvshmem"`: enables linking of the NVSHMEM device bitcode (`libnvshmem_device.bc`; for NVSHMEM 3.7 and later with per-SM bitcode the matching `.bc` is selected automatically) and registers the NVSHMEM cumodule init hook on the kernel so that device-side NVSHMEM calls initialize correctly. -- `compiler="clang"`: compiles the `.cu` source given by `file` with clang using `--cuda-device-only`. -- `file` / `extern_func_name`: the same as in the CUDA integration — the device-side source file and the function symbol inside it. - -Calling follows the CUDA integration, using `tle_raw.call` from a Triton kernel: - -```{code-block} python -@triton.jit -def simple_shift_kernel(destination_ptr): - tle_raw.call(simple_shift, [destination_ptr]) -``` - -The NVSHMEM-related host-side setup is provided by `triton.experimental.tle.raw.nvshmem.utils`: - -- `init_torch_distributed()` / `init_nvshmem_by_torch_pg(common, group)`: initialize torch.distributed and NVSHMEM from a PyTorch process group. -- `load_host(source)` / `load_common_host(source=None)`: compile and load a host-side `.cu` (with `nvshmem.h` and `nvshmemx.h`) and return a library object whose `extern "C"` functions can be called directly. -- `tensor_from_pointer(pointer, shape, dtype, device)`: create a non-owning Torch tensor view over CUDA memory. -- `copy_on_stream(dst_ptr, src_ptr, nbytes, stream)` and `set_signal_cuda_ptr(signal_ptr, signal, stream)`: on-stream copies and signal pointer setup. -- `print_perf(...)` / `print_perf_mean(...)`: aggregate and print performance data per rank. -- `enable_nvshmem_device_bc(enabled=True)` / `is_nvshmem_device_bc_enabled()` / `get_nvshmem_extern_libs(arch=None)`: device bitcode linking switch and external-library query (`library="nvshmem"` enables it automatically). - -Prerequisites: - -- Set the `NVSHMEM_HOME` environment variable (or `triton.knobs.nvidia.nvshmem_home`) so that `libnvshmem_host.so` and the device bitcode can be resolved. -- The examples target sm90 and newer and require `world_size >= 2`; multi-GPU launches pass the topology through `RANK`, `LOCAL_RANK`, `WORLD_SIZE`, and `LOCAL_WORLD_SIZE`. - -Full examples are in `python/tutorials/tle/raw/nvshmem/`: - -- `01-simple-shift`: the minimal NVSHMEM device-side put example, showing the dialect declaration and how host and device sides cooperate. -- `02-allgather-gemm`: all-gather GEMM over NVSHMEM (with a benchmark). -- `03-gemm-allreduce`: GEMM plus all-reduce over NVSHMEM multimem. -- `04-cuda-ipc-allreduce`: all-reduce over CUDA IPC (with a benchmark), using the CUDA dialect without `library="nvshmem"`. - -### Processing flow - -#### Frontend: CUDA-LLVM integration into Triton frontend and runtime - -| Step | Module | Key Pass Development | -|---|---|---| -| Dialect registration entry (dialect decorator) | `python/triton/experimental/tle/raw/runtime.py` | - Maintains `registry = {"cuda": CUDAJITFunction, "mlir": MLIRJITFunction}`.
- `dialect(name="cuda", ...)` constructs a `CUDAJITFunction` object. | -| TTIR extension: `Tle_DSLRegionOp` | `FlagTree/third_party/tle/dialect/include/IR/TleOps.td` | - Accepts Triton parameters;
- Wraps LLVM IR into the region field. | -| CUDA runtime: where clang is actually invoked | `python/triton/experimental/tle/raw/cuda/runtime.py` | - `CUDAJITFunction` reads the `.cu` source text at initialization.
- `make_llvm()` directly calls `subprocess.run(clang ...)` to produce LLVM IR.
- `parse_llvm_ir(...)` converts the text into a module that can be plugged into the Triton builder.
![alt text](../assets/images/make_llvm.PNG) | - - -#### Middle-End: Python to C++, MLIR pass relationship and pass inheritance - -| Step | Module | Key Pass Development | -|---|---|---| -| Attach LLVM function to `dsl_region` | `python/triton/experimental/tle/language/raw/core.py` | - `call()` obtains the builder context.
- Triggers `func.make_llvm(context)`.
- Calls `create_tle_raw_region_by_llvm_func(...)` to generate the `dsl_region` op.
![alt text](../assets/images/dsl_region_op.PNG) | -| C++ bridge: IR injection and type bridging | `third_party/tle/triton_tle.cc` | - `third_party/tle/triton_tle.cc` exposes Python bindings for `create_tle_raw_region_by_llvm_func` and raw passes.
- `third_party/tle/triton_tle_raw.cc` implements `createTLERawRegionByLLVMFunc`: parses the function, clones it into the current module, performs parameter/return type mapping, and creates `tle::DSLRegionOp` + `tle::YieldOp`. | - - -#### Backend: CUDA-LLVM IR conversion — parameter handling and Triton/TLE-Raw data bridging - -| Step | Module | Key Pass Development | -|---|---|---| -| Backend pass registration | `third_party/nvidia/backend/compiler.py` | - `make_ttgir()` inserts `tle.raw_passes.add_tle_convert_arg_to_memdesc(pm)`, which converts `dsl_region` tensor parameters into memdesc form.
- `make_llir()` inserts `tle.raw_passes.add_tle_dsl_region_inline(pm)`, which inlines `dsl_region` into the main control flow before LLVM conversion.
![alt text](../assets/images/make_ttgir_and_make_llir.png) | -| Parameter bridging | `TleConvertArgToMemDesc` (TTGIR stage) | - Converts tensor parameters/results in `dsl_region` to memdesc semantics, adding local storage and synchronization.
- Key actions:
  - tensor operand → `LocalAlloc` + `LocalStore`;
  - `dsl_region` result tensor → `LocalLoad` readback;
  - inserts `NVVM::Barrier0Op` when necessary;
  - handles pack-related new types. | -| LLVM inline preparation | `TleDSLRegionInline` (LLIR stage) | - Inlines `tle.dsl_region` from a region op.
- Key actions:
  - splits block, creates continuation;
  - rewrites yield as a branch to continuation;
  - replaces original `dsl_region` result uses;
  - erases `dsl_region` op. | - - -#### Semantic object mapping - -| Semantic Object | Triton Side | TLE-Raw Side | LLVM Side | -|---|---|---|---| -| Scalar parameter | `i32`/`f32`/... | Directly as `Tle_ArgType` | LLVM scalar parameter | -| Pointer parameter | `tt.ptr` | Passed directly or extracted as LLVM ptr | `attribute((address_space(N))) T*` | -| Tensor input | `tensor<...>` | Converted to `ttg.memdesc` / `dsl_region` operand | Expanded to allocated/aligned/offset/sizes/strides | -| Tensor output | `tensor<...>` | `dsl_region` result + `tle.pack`/yield | LLVM struct or multi-return fields, then repacked | +# Use TLE-Raw + +This section introduces how to use TLE-Raw. TLE-Raw is available on trition_3.6.x branch. + +TLE Raw provides a low-level extension interface for Triton, allowing users to fill capability gaps and gain fine-grained control via third-party dialects and languages (for example, using CUDA for thread-level scheduling, synchronization, and memory access). Users can choose between portability and composable optimization (via MLIR dialect integration) and maximum fine-grained control (via CUDA integration) based on the target hardware and toolchain maturity. + +## Integrate MLIR dialect into LLVM for portability and composable optimization + +The following is an example of MLIR (Multi-Level Intermediate Representation). + +```{code-block} python +from typing_extensions import Literal as L + +from mlir import ir +from mlir.dialects import arith, llvm, nvvm, scf +import torch +import triton +import triton.language as tl +from triton.experimental.tle.raw import dialect, Input +import triton.experimental.tle.language.raw as tle_raw + +DEVICE = triton.runtime.driver.active.get_active_torch_device() + +@dialect(name="mlir") +def vector_add_tile( + output: Input[L["!llvm.ptr<1>"]], + x: Input[L["!llvm.ptr<1>"]], + y: Input[L["!llvm.ptr<1>"]], + n_elements: Input[L["i32"]], +): + tidx = nvvm.read_ptx_sreg_tid_x(ir.IntegerType.get_signless(32)) + bdimx = nvvm.read_ptx_sreg_ntid_x(ir.IntegerType.get_signless(32)) + gdimx = nvvm.read_ptx_sreg_nctaid_x(ir.IntegerType.get_signless(32)) + bidx = nvvm.read_ptx_sreg_ctaid_x(ir.IntegerType.get_signless(32)) + tidx = arith.index_cast(ir.IndexType.get(), tidx) + bdimx = arith.index_cast(ir.IndexType.get(), bdimx) + gdimx = arith.index_cast(ir.IndexType.get(), gdimx) + bidx = arith.index_cast(ir.IndexType.get(), bidx) + idx = arith.addi(arith.muli(bidx, bdimx), tidx) + step = arith.muli(bdimx, gdimx) + n_elements = arith.index_cast(ir.IndexType.get(), n_elements) + for i in scf.for_(idx, n_elements, step): + i = arith.index_cast(ir.IntegerType.get_signless(32), i) + ptrty = ir.Type.parse("!llvm.ptr<1>") + f32ty = ir.Type.parse("f32") + xptr = llvm.getelementptr(ptrty, x, [i], [-2147483648], f32ty, 0) + yptr = llvm.getelementptr(ptrty, y, [i], [-2147483648], f32ty, 0) + xval = llvm.load(f32ty, xptr) + yval = llvm.load(f32ty, yptr) + outval = arith.addf(xval, yval) + outptr = llvm.getelementptr(ptrty, output, [i], [-2147483648], f32ty, 0) + llvm.store(outval, outptr) + scf.yield_([]) + +@triton.jit +def add_kernel( + x_ptr, + y_ptr, + output_ptr, + n_elements, + BLOCK_SIZE: tl.constexpr, +): + tle_raw.call(vector_add_tile, [output_ptr, x_ptr, y_ptr, n_elements]) + +def add(x: torch.Tensor, y: torch.Tensor): + output = torch.empty_like(x) + assert x.device == DEVICE and y.device == DEVICE and output.device == DEVICE + n_elements = output.numel() + grid = lambda meta: (triton.cdiv(n_elements, meta["BLOCK_SIZE"]), ) + add_kernel[grid](x, y, output, n_elements, BLOCK_SIZE=1024) + return output + +if __name__ == "__main__": + x = torch.randn(2048, device=DEVICE) + y = torch.randn(2048, device=DEVICE) + z = add(x, y) + assert torch.allclose(x + y, z), (x + y, z) +``` + +TLE-raw consists of the following parts: + +- Dialect declaration (decorator) + - Decorator: @tle.raw.language(name="mlir") + - Explanation: This decorator marks the function vector_add_tile as a block of code written directly in the MLIR dialect. It tells the compiler, specifically through the FlagTree EDSL (Embedded Domain Specific Language), that the body of this function should be interpreted and lowered using MLIR operations (such as nvvm, arith, and tensor), rather than standard Python or Triton operations. +- Function implementation + - Function: vector_add_tile(...) + - Explanation: This is the actual implementation of the computation kernel written using low-level MLIR Python bindings. It defines the specific operations (thread indexing, memory loading, floating-point addition, and memory storing) that will be executed by the hardware. +- Function call + - Invocation: tle_raw.call(vector_add_tile, args=[x, y, output]) + - Explanation: This line invokes the declared MLIR function (vector_add_tile) from within the high-level Triton kernel (add_kernel). It passes the input tensors x, y, and the output buffer. Crucially, it provides hardware mapping hints (defining the number of threads) and memory layout specifications (defining the tensors as residing in "shared" memory with a specific order). This allows the compiler to bridge the gap between the high-level tl.load/tl.store operations and the low-level MLIR IR generation. + +## Integrate CUDA into LLVM for maximum fine-grained control + +This section only covers how to integrate CUDA kernel into LLVM inline path. Integrating other vendors into LLVM inline path can follow similar steps. + +TLE-Raw supports CUDA kernel integration via the LLVM inline path. Vendors integrating TLE-Raw on the CUDA side should evaluate: + +- Whether clang can generate LLVM IR and serialize it as text +- Whether TTGIR-related pass operations can be reused or adapted + + +### LLVM route + +Basic flow: use clang to translate CUDA code into LLVM IR, then apply the existing LLVM inline pass. + +![alt text](../assets/images/cuda-to-vllm-line-pass.png) + +### Usage example + +- Reference: `python/tutorials/tle/raw/cuda/01-vector-add.py` + +- Triton side: provide the CUDA file path and function declaration. Other vendors could register your own language name as the value of the `name` parameter. + ![alt text](../assets/images/triton-side.png) + + +- CUDA side: implement the CUDA kernel. LLVM struct parameter declarations are still retained (because subsequent inlining requires handling the Triton ptr-to-LLVM conversion, which is currently left to the user for one-to-one mapping). Other vendors should customize the mapping based on your own language. + + ![alt text](../assets/images/cuda-side.PNG) + + +### Multi-GPU communication with NVSHMEM + +NVSHMEM is NVIDIA's PGAS (Partitioned Global Address Space) communication library for multi-GPU and multi-node programs. It gives every rank a unified address space: a region of memory on each GPU (symmetric memory) is reachable both from the local GPU and directly from other GPUs, so cross-GPU communication can be expressed with ordinary load/store or put/get semantics instead of hand-written point-to-point send/receive logic. It provides both host-side and device-side APIs, and the device-side interfaces can be called from inside a kernel, which makes it a natural fit for all-gather, all-reduce, and fused GEMM-plus-communication patterns. NVSHMEM is used only for NVIDIA GPUs; the distributed primitives of TLE-Lite and TLE-Struct use FlagCX instead of NVSHMEM. + +On top of the CUDA integration, `@dialect` adds the `library` and `compiler` parameters so that NVSHMEM device-side interfaces can be inlined directly into a TLE-Raw kernel: + +```{code-block} python +@dialect( + name="cuda", + library="nvshmem", + compiler="clang", + file=(Path(__file__).parent / "simple-shift-device.cu"), + extern_func_name="simple_shift", +) +def simple_shift(*args, **kwargs): + ... +``` + +- `library="nvshmem"`: enables linking of the NVSHMEM device bitcode (`libnvshmem_device.bc`; for NVSHMEM 3.7 and later with per-SM bitcode the matching `.bc` is selected automatically) and registers the NVSHMEM cumodule init hook on the kernel so that device-side NVSHMEM calls initialize correctly. +- `compiler="clang"`: compiles the `.cu` source given by `file` with clang using `--cuda-device-only`. +- `file` / `extern_func_name`: the same as in the CUDA integration — the device-side source file and the function symbol inside it. + +Calling follows the CUDA integration, using `tle_raw.call` from a Triton kernel: + +```{code-block} python +@triton.jit +def simple_shift_kernel(destination_ptr): + tle_raw.call(simple_shift, [destination_ptr]) +``` + +The NVSHMEM-related host-side setup is provided by `triton.experimental.tle.raw.nvshmem.utils`: + +- `init_torch_distributed()` / `init_nvshmem_by_torch_pg(common, group)`: initialize torch.distributed and NVSHMEM from a PyTorch process group. +- `load_host(source)` / `load_common_host(source=None)`: compile and load a host-side `.cu` (with `nvshmem.h` and `nvshmemx.h`) and return a library object whose `extern "C"` functions can be called directly. +- `tensor_from_pointer(pointer, shape, dtype, device)`: create a non-owning Torch tensor view over CUDA memory. +- `copy_on_stream(dst_ptr, src_ptr, nbytes, stream)` and `set_signal_cuda_ptr(signal_ptr, signal, stream)`: on-stream copies and signal pointer setup. +- `print_perf(...)` / `print_perf_mean(...)`: aggregate and print performance data per rank. +- `enable_nvshmem_device_bc(enabled=True)` / `is_nvshmem_device_bc_enabled()` / `get_nvshmem_extern_libs(arch=None)`: device bitcode linking switch and external-library query (`library="nvshmem"` enables it automatically). + +Prerequisites: + +- Set the `NVSHMEM_HOME` environment variable (or `triton.knobs.nvidia.nvshmem_home`) so that `libnvshmem_host.so` and the device bitcode can be resolved. +- The examples target sm90 and newer and require `world_size >= 2`; multi-GPU launches pass the topology through `RANK`, `LOCAL_RANK`, `WORLD_SIZE`, and `LOCAL_WORLD_SIZE`. + +Full examples are in `python/tutorials/tle/raw/nvshmem/`: + +- `01-simple-shift`: the minimal NVSHMEM device-side put example, showing the dialect declaration and how host and device sides cooperate. +- `02-allgather-gemm`: all-gather GEMM over NVSHMEM (with a benchmark). +- `03-gemm-allreduce`: GEMM plus all-reduce over NVSHMEM multimem. +- `04-cuda-ipc-allreduce`: all-reduce over CUDA IPC (with a benchmark). + +### Processing flow + +#### Frontend: CUDA-LLVM integration into Triton frontend and runtime + +| Step | Module | Key Pass Development | +|---|---|---| +| Dialect registration entry (dialect decorator) | `python/triton/experimental/tle/raw/runtime.py` | - Maintains `registry = {"cuda": CUDAJITFunction, "mlir": MLIRJITFunction}`.
- `dialect(name="cuda", ...)` constructs a `CUDAJITFunction` object. | +| TTIR extension: `Tle_DSLRegionOp` | `FlagTree/third_party/tle/dialect/include/IR/TleOps.td` | - Accepts Triton parameters;
- Wraps LLVM IR into the region field. | +| CUDA runtime: where clang is actually invoked | `python/triton/experimental/tle/raw/cuda/runtime.py` | - `CUDAJITFunction` reads the `.cu` source text at initialization.
- `make_llvm()` directly calls `subprocess.run(clang ...)` to produce LLVM IR.
- `parse_llvm_ir(...)` converts the text into a module that can be plugged into the Triton builder.
![alt text](../assets/images/make_llvm.PNG) | + + +#### Middle-End: Python to C++, MLIR pass relationship and pass inheritance + +| Step | Module | Key Pass Development | +|---|---|---| +| Attach LLVM function to `dsl_region` | `python/triton/experimental/tle/language/raw/core.py` | - `call()` obtains the builder context.
- Triggers `func.make_llvm(context)`.
- Calls `create_tle_raw_region_by_llvm_func(...)` to generate the `dsl_region` op.
![alt text](../assets/images/dsl_region_op.PNG) | +| C++ bridge: IR injection and type bridging | `third_party/tle/triton_tle.cc` | - `third_party/tle/triton_tle.cc` exposes Python bindings for `create_tle_raw_region_by_llvm_func` and raw passes.
- `third_party/tle/triton_tle_raw.cc` implements `createTLERawRegionByLLVMFunc`: parses the function, clones it into the current module, performs parameter/return type mapping, and creates `tle::DSLRegionOp` + `tle::YieldOp`. | + + +#### Backend: CUDA-LLVM IR conversion — parameter handling and Triton/TLE-Raw data bridging + +| Step | Module | Key Pass Development | +|---|---|---| +| Backend pass registration | `third_party/nvidia/backend/compiler.py` | - `make_ttgir()` inserts `tle.raw_passes.add_tle_convert_arg_to_memdesc(pm)`, which converts `dsl_region` tensor parameters into memdesc form.
- `make_llir()` inserts `tle.raw_passes.add_tle_dsl_region_inline(pm)`, which inlines `dsl_region` into the main control flow before LLVM conversion.
![alt text](../assets/images/make_ttgir_and_make_llir.png) | +| Parameter bridging | `TleConvertArgToMemDesc` (TTGIR stage) | - Converts tensor parameters/results in `dsl_region` to memdesc semantics, adding local storage and synchronization.
- Key actions:
  - tensor operand → `LocalAlloc` + `LocalStore`;
  - `dsl_region` result tensor → `LocalLoad` readback;
  - inserts `NVVM::Barrier0Op` when necessary;
  - handles pack-related new types. | +| LLVM inline preparation | `TleDSLRegionInline` (LLIR stage) | - Inlines `tle.dsl_region` from a region op.
- Key actions:
  - splits block, creates continuation;
  - rewrites yield as a branch to continuation;
  - replaces original `dsl_region` result uses;
  - erases `dsl_region` op. | + + +#### Semantic object mapping + +| Semantic Object | Triton Side | TLE-Raw Side | LLVM Side | +|---|---|---|---| +| Scalar parameter | `i32`/`f32`/... | Directly as `Tle_ArgType` | LLVM scalar parameter | +| Pointer parameter | `tt.ptr` | Passed directly or extracted as LLVM ptr | `attribute((address_space(N))) T*` | +| Tensor input | `tensor<...>` | Converted to `ttg.memdesc` / `dsl_region` operand | Expanded to allocated/aligned/offset/sizes/strides | +| Tensor output | `tensor<...>` | `dsl_region` result + `tle.pack`/yield | LLVM struct or multi-return fields, then repacked | diff --git a/docs/flagtree_zh/release_notes/release_notes_v070.md b/docs/flagtree_zh/release_notes/release_notes_v070.md index a8288acd3f..82b0abc291 100644 --- a/docs/flagtree_zh/release_notes/release_notes_v070.md +++ b/docs/flagtree_zh/release_notes/release_notes_v070.md @@ -1,28 +1,27 @@ -# FlagTree 0.7.0 发布 - -- **新增特性** - - 3.6.x 分支: - - TLE-Lite: - - 新增 `tle.shard_id` 操作,用于查询当前 program 在 device mesh 指定轴上的坐标。在 NVIDIA 上支持。 - - 新增以下分布式操作:`tle.signal` 和 `tle.signal_wait`。在 NVIDIA 上支持。 - - 为 `tle.remote` 扩展 `space`(cluster / device / node)、`dtype`、`offset`、`coopkind` 和 `netidx` 参数,覆盖线程块 cluster 内(DSMEM)、节点内 GPU 之间(NVLink P2P)以及跨节点(FlagCX/RDMA)的远程访问。在 NVIDIA 上支持。 - - 为 `tle.distributed_barrier` 扩展 `space`、`group_kind`、`barrier_kind` 和 `order` 参数。在 NVIDIA 上支持。 - - TLE-Struct: - - 通过 `tle.gpu.alloc(..., alias=...)` 新增 GPU 缓冲区别名(buffer aliasing),提供带类型的共享内存视图,并对别名视图进行静态校验。 - - 新增 `tle.gpu.set_layout` 操作,用于显式分布式布局赋值,并提供 `BlockEncoding`、`MmaEncoding`、`DotOperandEncoding` 和 `SlicedEncoding` 布局对象。 - - 新增以下 barrier 操作:`tle.gpu.alloc_barrier`、`tle.gpu.alloc_barriers`、`tle.gpu.barrier_wait` 和 `tle.gpu.barrier_arrive`;为 `tle.gpu.copy` 新增 `barrier` 和 `mask` 参数。在 NVIDIA 上支持。 - - 新增 `tle.gpu.wgmma` 和 `tle.gpu.wgmma_wait` 操作。在 NVIDIA 上支持。 - - 新增 `tle.gpu.buffered_tensor.slot` 和 `tle.gpu.buffered_tensor.reshape` 操作。在 NVIDIA 上支持。 - - 为 `tle.gpu.alloc` 新增 `init_value` 和 `alias_offset_bytes` 参数。在 NVIDIA 上支持。 - - TLE-Raw: - - 新增 `library` 与 `compiler` 参数,支持通过 `@dialect(..., library="nvshmem", compiler="clang")` 将 NVSHMEM 设备端接口内联进 TLE-Raw kernel。在 NVIDIA 上支持。 - - - 后端: - - 新增以下后端集成(基于 Triton 3.6)并新增 CI/CD:[tileir](/getting_started/multi-backend-prebuilt-docker-image-install/install-tileir.md)(NVIDIA TileIR)、[ppu](/getting_started/multi-backend-prebuilt-docker-image-install/install-ppu.md)(平头哥)和 [spacemit](/getting_started/multi-backend-prebuilt-docker-image-install/install-spacemit.md)(进迭时空)。 - - 将以下后端升级至 Triton 3.6 并新增 CI/CD:[sunrise](/getting_started/multi-backend-prebuilt-docker-image-install/install-sunrise.md)、[xpu](/getting_started/multi-backend-prebuilt-docker-image-install/install-xpu.md)、[iluvatar](/getting_started/multi-backend-prebuilt-docker-image-install/install-iluvatar.md) 和 [tsingmicro](/getting_started/multi-backend-prebuilt-docker-image-install/install-tsingmicro.md)。 - - 为 [amd](/getting_started/multi-backend-prebuilt-docker-image-install/install-amd.md) 后端新增 TLE 支持并新增 CI/CD。 - - 另外还支持 [rpu](/getting_started/multi-backend-prebuilt-docker-image-install/install-rpu.md)(辉羲智能,Triton 3.6)。在 3.3.x 分支上,[ARM64 CPU](/getting_started/install-arm64-cpu.md) 提供 [TLE-CPU](/user_guide/use-tle-cpu.md)。 - - [ARM64 CPU](/getting_started/flagtree-cpu.md) 基于 Triton 3.7.2 提供 CPU 后端,已在 Linux Arm64 上通过 vector-add kernel 与 FlagGems W4A8 算子测试验证。 - -- **DevTools(调试器与性能分析器)** - - FlagPrism([flagos-ai/FlagPrism](https://github.com/flagos-ai/FlagPrism))为 Triton 程序提供调试与性能分析工具,包含 `flagtree.debugger` 与 `flagtree.profiler`,以 `third_party/FlagPrism` 子模块集成在 FlagTree 中。先支持部分后端:华为昇腾、天数智芯、摩尔线程。 +# FlagTree 0.7.0 发布 + +- **新增特性** + - 3.6.x 分支: + - TLE-Lite: + - 新增 `tle.shard_id` 操作,用于查询当前 program 在 device mesh 指定轴上的坐标。在 NVIDIA 上支持。 + - 新增以下分布式操作:`tle.signal` 和 `tle.signal_wait`。在 NVIDIA 上支持。 + - 为 `tle.remote` 扩展 `space`(cluster / device / node)、`dtype`、`offset`、`coopkind` 和 `netidx` 参数,覆盖线程块 cluster 内(DSMEM)、节点内 GPU 之间(NVLink P2P)以及跨节点(FlagCX/RDMA)的远程访问。在 NVIDIA 上支持。 + - 为 `tle.distributed_barrier` 扩展 `space`、`group_kind`、`barrier_kind` 和 `order` 参数。在 NVIDIA 上支持。 + - TLE-Struct: + - 通过 `tle.gpu.alloc(..., alias=...)` 新增 GPU 缓冲区别名(buffer aliasing),提供带类型的共享内存视图,并对别名视图进行静态校验。 + - 新增 `tle.gpu.set_layout` 操作,用于显式分布式布局赋值,并提供 `BlockEncoding`、`MmaEncoding`、`DotOperandEncoding` 和 `SlicedEncoding` 布局对象。 + - 新增以下 barrier 操作:`tle.gpu.alloc_barrier`、`tle.gpu.alloc_barriers`、`tle.gpu.barrier_wait` 和 `tle.gpu.barrier_arrive`;为 `tle.gpu.copy` 新增 `barrier` 和 `mask` 参数。在 NVIDIA 上支持。 + - 新增 `tle.gpu.wgmma` 和 `tle.gpu.wgmma_wait` 操作。在 NVIDIA 上支持。 + - 新增 `tle.gpu.buffered_tensor.slot` 和 `tle.gpu.buffered_tensor.reshape` 操作。在 NVIDIA 上支持。 + - 为 `tle.gpu.alloc` 新增 `init_value` 和 `alias_offset_bytes` 参数。在 NVIDIA 上支持。 + - TLE-Raw: + - 新增 `library` 与 `compiler` 参数,支持通过 `@dialect(..., library="nvshmem", compiler="clang")` 将 NVSHMEM 设备端接口内联进 TLE-Raw kernel。在 NVIDIA 上支持。 + + - 后端: + - 新增以下后端集成(基于 Triton 3.6):[tileir](/getting_started/multi-backend-prebuilt-docker-image-install/install-tileir.md)(NVIDIA TileIR)、[ppu](/getting_started/multi-backend-prebuilt-docker-image-install/install-ppu.md)(平头哥)和 [spacemit](/getting_started/multi-backend-prebuilt-docker-image-install/install-spacemit.md)(进迭时空)。 + - 将以下后端升级至 Triton 3.6:[sunrise](/getting_started/multi-backend-prebuilt-docker-image-install/install-sunrise.md)、[xpu](/getting_started/multi-backend-prebuilt-docker-image-install/install-xpu.md)、[iluvatar](/getting_started/multi-backend-prebuilt-docker-image-install/install-iluvatar.md) 和 [tsingmicro](/getting_started/multi-backend-prebuilt-docker-image-install/install-tsingmicro.md)。 + - 为 [amd](/getting_started/multi-backend-prebuilt-docker-image-install/install-amd.md) 后端新增 TLE 支持。 + - [ARM64 CPU](/getting_started/flagtree-cpu.md) 基于 Triton 3.7.2 提供 CPU 后端,已在 Linux Arm64 上通过 vector-add kernel 与 FlagGems W4A8 算子测试验证。 + +- **DevTools(调试器与性能分析器)** + - FlagPrism([flagos-ai/FlagPrism](https://github.com/flagos-ai/FlagPrism))为 Triton 程序提供调试与性能分析工具,包含 `flagtree.debugger` 与 `flagtree.profiler`,以 `third_party/FlagPrism` 子模块集成在 FlagTree 中。先支持部分后端:华为昇腾、天数智芯、摩尔线程。 diff --git a/docs/flagtree_zh/user_guide/use-tle-raw.md b/docs/flagtree_zh/user_guide/use-tle-raw.md index b3a25d42d9..c0dd7e87fa 100644 --- a/docs/flagtree_zh/user_guide/use-tle-raw.md +++ b/docs/flagtree_zh/user_guide/use-tle-raw.md @@ -1,207 +1,207 @@ -# 使用 TLE-Raw - -本节介绍如何使用 TLE-Raw。TLE-Raw 在 trition_3.6.x 分支上可用。 - -TLE Raw 为 Triton 提供了低级扩展接口,允许用户通过第三方方言和语言(例如使用 CUDA 进行线程级调度、同步和内存访问)来填补能力空白并获得细粒度控制。用户可以根据目标硬件和工具链成熟度,在可移植性和可组合优化(通过 MLIR 方言集成)与最大细粒度控制(通过 CUDA 集成)之间进行选择。 - -## 将 MLIR 方言集成到 LLVM 中以实现可移植性和可组合优化 - -以下是 MLIR(多级中间表示)的示例。 - -```{code-block} python -from typing_extensions import Literal as L - -from mlir import ir -from mlir.dialects import arith, llvm, nvvm, scf -import torch -import triton -import triton.language as tl -from triton.experimental.tle.raw import dialect, Input -import triton.experimental.tle.language.raw as tle_raw - -DEVICE = triton.runtime.driver.active.get_active_torch_device() - -@dialect(name="mlir") -def vector_add_tile( - output: Input[L["!llvm.ptr<1>"]], - x: Input[L["!llvm.ptr<1>"]], - y: Input[L["!llvm.ptr<1>"]], - n_elements: Input[L["i32"]], -): - tidx = nvvm.read_ptx_sreg_tid_x(ir.IntegerType.get_signless(32)) - bdimx = nvvm.read_ptx_sreg_ntid_x(ir.IntegerType.get_signless(32)) - gdimx = nvvm.read_ptx_sreg_nctaid_x(ir.IntegerType.get_signless(32)) - bidx = nvvm.read_ptx_sreg_ctaid_x(ir.IntegerType.get_signless(32)) - tidx = arith.index_cast(ir.IndexType.get(), tidx) - bdimx = arith.index_cast(ir.IndexType.get(), bdimx) - gdimx = arith.index_cast(ir.IndexType.get(), gdimx) - bidx = arith.index_cast(ir.IndexType.get(), bidx) - idx = arith.addi(arith.muli(bidx, bdimx), tidx) - step = arith.muli(bdimx, gdimx) - n_elements = arith.index_cast(ir.IndexType.get(), n_elements) - for i in scf.for_(idx, n_elements, step): - i = arith.index_cast(ir.IntegerType.get_signless(32), i) - ptrty = ir.Type.parse("!llvm.ptr<1>") - f32ty = ir.Type.parse("f32") - xptr = llvm.getelementptr(ptrty, x, [i], [-2147483648], f32ty, 0) - yptr = llvm.getelementptr(ptrty, y, [i], [-2147483648], f32ty, 0) - xval = llvm.load(f32ty, xptr) - yval = llvm.load(f32ty, yptr) - outval = arith.addf(xval, yval) - outptr = llvm.getelementptr(ptrty, output, [i], [-2147483648], f32ty, 0) - llvm.store(outval, outptr) - scf.yield_([]) - -@triton.jit -def add_kernel( - x_ptr, - y_ptr, - output_ptr, - n_elements, - BLOCK_SIZE: tl.constexpr, -): - tle_raw.call(vector_add_tile, [output_ptr, x_ptr, y_ptr, n_elements]) - -def add(x: torch.Tensor, y: torch.Tensor): - output = torch.empty_like(x) - assert x.device == DEVICE and y.device == DEVICE and output.device == DEVICE - n_elements = output.numel() - grid = lambda meta: (triton.cdiv(n_elements, meta["BLOCK_SIZE"]), ) - add_kernel[grid](x, y, output, n_elements, BLOCK_SIZE=1024) - return output - -if __name__ == "__main__": - x = torch.randn(2048, device=DEVICE) - y = torch.randn(2048, device=DEVICE) - z = add(x, y) - assert torch.allclose(x + y, z), (x + y, z) -``` - -TLE-raw 由以下部分组成: - -- 方言声明(装饰器) - - 装饰器: @tle.raw.language(name="mlir") - - 说明: 此装饰器将函数 vector_add_tile 标记为直接用 MLIR 方言编写的代码块。它告诉编译器(具体通过 FlagTree EDSL(嵌入式领域特定语言)),该函数体应使用 MLIR 操作(如 nvvm、arith 和 tensor)来解释和下层,而不是标准的 Python 或 Triton 操作。 -- 函数实现 - - 函数: vector_add_tile(...) - - 说明: 这是使用低级 MLIR Python 绑定编写的计算内核的实际实现。它定义了将由硬件执行的具体操作(线程索引、内存加载、浮点加法和内存存储)。 -- 函数调用 - - 调用: tle_raw.call(vector_add_tile, args=[x, y, output]) - - 说明: 此行从高级 Triton 内核(add_kernel)中调用已声明的 MLIR 函数(vector_add_tile)。它传递输入张量 x、y 和输出缓冲区。关键是,它提供了硬件映射提示(定义线程数量)和内存布局规范(定义张量驻留在"shared"内存中并具有特定顺序)。这使得编译器能够弥合高级 tl.load/tl.store 操作与低级 MLIR IR 生成之间的差距。 - -## 将 CUDA 集成到 LLVM 中以实现最大细粒度控制 - -本节仅介绍如何将 CUDA 内核集成到 LLVM 内联路径中。将其他厂商集成到 LLVM 内联路径中可以遵循类似的步骤。 - -TLE-Raw 通过 LLVM 内联路径支持 CUDA 内核集成。在 CUDA 侧集成 TLE-Raw 的厂商应评估: - -- clang 是否能生成 LLVM IR 并将其序列化为文本 -- TTGIR 相关的 pass 操作是否可以重用或适配 - - -### LLVM 路线 - -基本流程:使用 clang 将 CUDA 代码转换为 LLVM IR,然后应用现有的 LLVM 内联 pass。 - -![alt text](../assets/images/cuda-to-vllm-line-pass.png) - -### 使用示例 - -- 参考: `python/tutorials/tle/raw/cuda/01-vector-add.py` - -- Triton 侧: 提供 CUDA 文件路径和函数声明。其他厂商可以注册自己的语言名称作为 `name` 参数的值。 - ![alt text](../assets/images/triton-side.png) - - -- CUDA 侧: 实现 CUDA 内核。LLVM 结构体参数声明仍然保留(因为后续内联需要处理 Triton ptr 到 LLVM 的转换,目前留给用户进行一对一映射)。其他厂商应根据自己的语言自定义映射。 - - ![alt text](../assets/images/cuda-side.PNG) - - -### 使用 NVSHMEM 进行多卡通信 - -NVSHMEM 是 NVIDIA 提供的 PGAS(Partitioned Global Address Space)通信库,为多卡/多节点程序提供统一地址空间:每张 GPU 上的一块内存(对称内存)既可以被本卡访问,也可以被其他 GPU 直接访问,因此跨卡通信可以用普通的 load/store 或 put/get 语义表达,而不必显式编写点对点收发逻辑。它提供主机端与设备端两套 API,设备端接口可在 kernel 内部直接调用,常用于 all-gather、all-reduce、GEMM + 通信融合等场景。NVSHMEM 仅用于 NVIDIA 卡;TLE-Lite 和 TLE-Struct 的分布式原语走的是 FlagCX,不使用 NVSHMEM。 - -`@dialect` 在 CUDA 集成的基础上新增 `library` 与 `compiler` 参数,用于把 NVSHMEM 设备端接口直接内联进 TLE-Raw kernel: - -```{code-block} python -@dialect( - name="cuda", - library="nvshmem", - compiler="clang", - file=(Path(__file__).parent / "simple-shift-device.cu"), - extern_func_name="simple_shift", -) -def simple_shift(*args, **kwargs): - ... -``` - -- `library="nvshmem"`:启用 NVSHMEM 设备端 bitcode 链接(`libnvshmem_device.bc`,NVSHMEM 3.7 及以上按 SM 拆分时自动选择对应 `.bc`),并在 kernel 上注册 NVSHMEM cumodule init hook,使设备端 NVSHMEM 调用可以正常初始化。 -- `compiler="clang"`:使用 clang 以 `--cuda-device-only` 编译 `file` 指定的 `.cu` 源文件。 -- `file` / `extern_func_name`:与 CUDA 集成一致,分别指定设备端源文件与其中的函数符号。 - -调用方式与 CUDA 集成相同,通过 `tle_raw.call` 从 Triton kernel 中调用: - -```{code-block} python -@triton.jit -def simple_shift_kernel(destination_ptr): - tle_raw.call(simple_shift, [destination_ptr]) -``` - -NVSHMEM 相关的 host 侧准备工作由 `triton.experimental.tle.raw.nvshmem.utils` 提供: - -- `init_torch_distributed()` / `init_nvshmem_by_torch_pg(common, group)`:基于 PyTorch 进程组初始化 torch.distributed 与 NVSHMEM。 -- `load_host(source)` / `load_common_host(source=None)`:编译并加载 host 侧 `.cu`(含 `nvshmem.h`、`nvshmemx.h`),返回可直接调用其中 `extern "C"` 函数的库对象。 -- `tensor_from_pointer(pointer, shape, dtype, device)`:在 CUDA 显存上创建非拥有的 Torch tensor 视图。 -- `copy_on_stream(dst_ptr, src_ptr, nbytes, stream)`、`set_signal_cuda_ptr(signal_ptr, signal, stream)`:流上的拷贝与 signal 指针设置。 -- `print_perf(...)` / `print_perf_mean(...)`:按 rank 汇总打印性能数据。 -- `enable_nvshmem_device_bc(enabled=True)` / `is_nvshmem_device_bc_enabled()` / `get_nvshmem_extern_libs(arch=None)`:设备端 bitcode 链接开关与外部库查询(`library="nvshmem"` 会自动开启)。 - -使用前需要满足: - -- 设置 `NVSHMEM_HOME` 环境变量(或通过 `triton.knobs.nvidia.nvshmem_home` 指定),用于解析 `libnvshmem_host.so` 与设备端 bitcode。 -- 示例面向 sm90 及以上架构,且要求 `world_size >= 2`;多卡启动时通过 `RANK`、`LOCAL_RANK`、`WORLD_SIZE`、`LOCAL_WORLD_SIZE` 传入拓扑。 - -完整示例见 `python/tutorials/tle/raw/nvshmem/`: - -- `01-simple-shift`:最小的 NVSHMEM 设备端 put 示例,演示 dialect 声明与 host/device 两侧的配合。 -- `02-allgather-gemm`:基于 NVSHMEM 的 all-gather GEMM(含 benchmark)。 -- `03-gemm-allreduce`:基于 NVSHMEM multimem 的 GEMM + all-reduce。 -- `04-cuda-ipc-allreduce`:基于 CUDA IPC 的 all-reduce(含 benchmark),使用不带 `library="nvshmem"` 的 CUDA 方言。 - -### 处理流程 - -#### 前端: CUDA-LLVM 集成到 Triton 前端和运行时 - -| 步骤 | 模块 | 关键 Pass 开发 | -|---|---|---| -| 方言注册入口(dialect 装饰器) | `python/triton/experimental/tle/raw/runtime.py` | - 维护 `registry = {"cuda": CUDAJITFunction, "mlir": MLIRJITFunction}`。
- `dialect(name="cuda", ...)` 构造 `CUDAJITFunction` 对象。 | -| TTIR 扩展: `Tle_DSLRegionOp` | `FlagTree/third_party/tle/dialect/include/IR/TleOps.td` | - 接受 Triton 参数;
- 将 LLVM IR 包装到 region 字段中。 | -| CUDA 运行时: 实际调用 clang 的位置 | `python/triton/experimental/tle/raw/cuda/runtime.py` | - `CUDAJITFunction` 在初始化时读取 `.cu` 源文本。
- `make_llvm()` 直接调用 `subprocess.run(clang ...)` 生成 LLVM IR。
- `parse_llvm_ir(...)` 将文本转换为可插入 Triton builder 的模块。
![alt text](../assets/images/make_llvm.PNG) | - - -#### 中端: Python 到 C++,MLIR pass 关系和 pass 继承 - -| 步骤 | 模块 | 关键 Pass 开发 | -|---|---|---| -| 将 LLVM 函数附加到 `dsl_region` | `python/triton/experimental/tle/language/raw/core.py` | - `call()` 获取 builder 上下文。
- 触发 `func.make_llvm(context)`。
- 调用 `create_tle_raw_region_by_llvm_func(...)` 生成 `dsl_region` op。
![alt text](../assets/images/dsl_region_op.PNG) | -| C++ 桥接: IR 注入和类型桥接 | `third_party/tle/triton_tle.cc` | - `third_party/tle/triton_tle.cc` 为 `create_tle_raw_region_by_llvm_func` 和 raw passes 暴露 Python 绑定。
- `third_party/tle/triton_tle_raw.cc` 实现 `createTLERawRegionByLLVMFunc`: 解析函数,将其克隆到当前模块,执行参数/返回类型映射,并创建 `tle::DSLRegionOp` + `tle::YieldOp`。 | - - -#### 后端: CUDA-LLVM IR 转换 — 参数处理和 Triton/TLE-Raw 数据桥接 - -| 步骤 | 模块 | 关键 Pass 开发 | -|---|---|---| -| 后端 pass 注册 | `third_party/nvidia/backend/compiler.py` | - `make_ttgir()` 插入 `tle.raw_passes.add_tle_convert_arg_to_memdesc(pm)`,将 `dsl_region` 张量参数转换为 memdesc 形式。
- `make_llir()` 插入 `tle.raw_passes.add_tle_dsl_region_inline(pm)`,在 LLVM 转换之前将 `dsl_region` 内联到主控制流中。
![alt text](../assets/images/make_ttgir_and_make_llir.png) | -| 参数桥接 | `TleConvertArgToMemDesc`(TTGIR 阶段) | - 将 `dsl_region` 中的张量参数/结果转换为 memdesc 语义,添加本地存储和同步。
- 关键操作:
  - 张量操作数 → `LocalAlloc` + `LocalStore`;
  - `dsl_region` 结果张量 → `LocalLoad` 读回;
  - 必要时插入 `NVVM::Barrier0Op`;
  - 处理 pack 相关的新类型。 | -| LLVM 内联准备 | `TleDSLRegionInline`(LLIR 阶段) | - 从 region op 内联 `tle.dsl_region`。
- 关键操作:
  - 拆分块,创建延续;
  - 将 yield 重写为到延续的分支;
  - 替换原始 `dsl_region` 结果的使用;
  - 删除 `dsl_region` op。 | - - -#### 语义对象映射 - -| 语义对象 | Triton 侧 | TLE-Raw 侧 | LLVM 侧 | -|---|---|---|---| -| 标量参数 | `i32`/`f32`/... | 直接作为 `Tle_ArgType` | LLVM 标量参数 | -| 指针参数 | `tt.ptr` | 直接传递或提取为 LLVM ptr | `attribute((address_space(N))) T*` | -| 张量输入 | `tensor<...>` | 转换为 `ttg.memdesc` / `dsl_region` 操作数 | 展开为 allocated/aligned/offset/sizes/strides | -| 张量输出 | `tensor<...>` | `dsl_region` 结果 + `tle.pack`/yield | LLVM 结构体或多返回字段,然后重新打包 | +# 使用 TLE-Raw + +本节介绍如何使用 TLE-Raw。TLE-Raw 在 trition_3.6.x 分支上可用。 + +TLE Raw 为 Triton 提供了低级扩展接口,允许用户通过第三方方言和语言(例如使用 CUDA 进行线程级调度、同步和内存访问)来填补能力空白并获得细粒度控制。用户可以根据目标硬件和工具链成熟度,在可移植性和可组合优化(通过 MLIR 方言集成)与最大细粒度控制(通过 CUDA 集成)之间进行选择。 + +## 将 MLIR 方言集成到 LLVM 中以实现可移植性和可组合优化 + +以下是 MLIR(多级中间表示)的示例。 + +```{code-block} python +from typing_extensions import Literal as L + +from mlir import ir +from mlir.dialects import arith, llvm, nvvm, scf +import torch +import triton +import triton.language as tl +from triton.experimental.tle.raw import dialect, Input +import triton.experimental.tle.language.raw as tle_raw + +DEVICE = triton.runtime.driver.active.get_active_torch_device() + +@dialect(name="mlir") +def vector_add_tile( + output: Input[L["!llvm.ptr<1>"]], + x: Input[L["!llvm.ptr<1>"]], + y: Input[L["!llvm.ptr<1>"]], + n_elements: Input[L["i32"]], +): + tidx = nvvm.read_ptx_sreg_tid_x(ir.IntegerType.get_signless(32)) + bdimx = nvvm.read_ptx_sreg_ntid_x(ir.IntegerType.get_signless(32)) + gdimx = nvvm.read_ptx_sreg_nctaid_x(ir.IntegerType.get_signless(32)) + bidx = nvvm.read_ptx_sreg_ctaid_x(ir.IntegerType.get_signless(32)) + tidx = arith.index_cast(ir.IndexType.get(), tidx) + bdimx = arith.index_cast(ir.IndexType.get(), bdimx) + gdimx = arith.index_cast(ir.IndexType.get(), gdimx) + bidx = arith.index_cast(ir.IndexType.get(), bidx) + idx = arith.addi(arith.muli(bidx, bdimx), tidx) + step = arith.muli(bdimx, gdimx) + n_elements = arith.index_cast(ir.IndexType.get(), n_elements) + for i in scf.for_(idx, n_elements, step): + i = arith.index_cast(ir.IntegerType.get_signless(32), i) + ptrty = ir.Type.parse("!llvm.ptr<1>") + f32ty = ir.Type.parse("f32") + xptr = llvm.getelementptr(ptrty, x, [i], [-2147483648], f32ty, 0) + yptr = llvm.getelementptr(ptrty, y, [i], [-2147483648], f32ty, 0) + xval = llvm.load(f32ty, xptr) + yval = llvm.load(f32ty, yptr) + outval = arith.addf(xval, yval) + outptr = llvm.getelementptr(ptrty, output, [i], [-2147483648], f32ty, 0) + llvm.store(outval, outptr) + scf.yield_([]) + +@triton.jit +def add_kernel( + x_ptr, + y_ptr, + output_ptr, + n_elements, + BLOCK_SIZE: tl.constexpr, +): + tle_raw.call(vector_add_tile, [output_ptr, x_ptr, y_ptr, n_elements]) + +def add(x: torch.Tensor, y: torch.Tensor): + output = torch.empty_like(x) + assert x.device == DEVICE and y.device == DEVICE and output.device == DEVICE + n_elements = output.numel() + grid = lambda meta: (triton.cdiv(n_elements, meta["BLOCK_SIZE"]), ) + add_kernel[grid](x, y, output, n_elements, BLOCK_SIZE=1024) + return output + +if __name__ == "__main__": + x = torch.randn(2048, device=DEVICE) + y = torch.randn(2048, device=DEVICE) + z = add(x, y) + assert torch.allclose(x + y, z), (x + y, z) +``` + +TLE-raw 由以下部分组成: + +- 方言声明(装饰器) + - 装饰器: @tle.raw.language(name="mlir") + - 说明: 此装饰器将函数 vector_add_tile 标记为直接用 MLIR 方言编写的代码块。它告诉编译器(具体通过 FlagTree EDSL(嵌入式领域特定语言)),该函数体应使用 MLIR 操作(如 nvvm、arith 和 tensor)来解释和下层,而不是标准的 Python 或 Triton 操作。 +- 函数实现 + - 函数: vector_add_tile(...) + - 说明: 这是使用低级 MLIR Python 绑定编写的计算内核的实际实现。它定义了将由硬件执行的具体操作(线程索引、内存加载、浮点加法和内存存储)。 +- 函数调用 + - 调用: tle_raw.call(vector_add_tile, args=[x, y, output]) + - 说明: 此行从高级 Triton 内核(add_kernel)中调用已声明的 MLIR 函数(vector_add_tile)。它传递输入张量 x、y 和输出缓冲区。关键是,它提供了硬件映射提示(定义线程数量)和内存布局规范(定义张量驻留在"shared"内存中并具有特定顺序)。这使得编译器能够弥合高级 tl.load/tl.store 操作与低级 MLIR IR 生成之间的差距。 + +## 将 CUDA 集成到 LLVM 中以实现最大细粒度控制 + +本节仅介绍如何将 CUDA 内核集成到 LLVM 内联路径中。将其他厂商集成到 LLVM 内联路径中可以遵循类似的步骤。 + +TLE-Raw 通过 LLVM 内联路径支持 CUDA 内核集成。在 CUDA 侧集成 TLE-Raw 的厂商应评估: + +- clang 是否能生成 LLVM IR 并将其序列化为文本 +- TTGIR 相关的 pass 操作是否可以重用或适配 + + +### LLVM 路线 + +基本流程:使用 clang 将 CUDA 代码转换为 LLVM IR,然后应用现有的 LLVM 内联 pass。 + +![alt text](../assets/images/cuda-to-vllm-line-pass.png) + +### 使用示例 + +- 参考: `python/tutorials/tle/raw/cuda/01-vector-add.py` + +- Triton 侧: 提供 CUDA 文件路径和函数声明。其他厂商可以注册自己的语言名称作为 `name` 参数的值。 + ![alt text](../assets/images/triton-side.png) + + +- CUDA 侧: 实现 CUDA 内核。LLVM 结构体参数声明仍然保留(因为后续内联需要处理 Triton ptr 到 LLVM 的转换,目前留给用户进行一对一映射)。其他厂商应根据自己的语言自定义映射。 + + ![alt text](../assets/images/cuda-side.PNG) + + +### 使用 NVSHMEM 进行多卡通信 + +NVSHMEM 是 NVIDIA 提供的 PGAS(Partitioned Global Address Space)通信库,为多卡/多节点程序提供统一地址空间:每张 GPU 上的一块内存(对称内存)既可以被本卡访问,也可以被其他 GPU 直接访问,因此跨卡通信可以用普通的 load/store 或 put/get 语义表达,而不必显式编写点对点收发逻辑。它提供主机端与设备端两套 API,设备端接口可在 kernel 内部直接调用,常用于 all-gather、all-reduce、GEMM + 通信融合等场景。NVSHMEM 仅用于 NVIDIA 卡;TLE-Lite 和 TLE-Struct 的分布式原语走的是 FlagCX,不使用 NVSHMEM。 + +`@dialect` 在 CUDA 集成的基础上新增 `library` 与 `compiler` 参数,用于把 NVSHMEM 设备端接口直接内联进 TLE-Raw kernel: + +```{code-block} python +@dialect( + name="cuda", + library="nvshmem", + compiler="clang", + file=(Path(__file__).parent / "simple-shift-device.cu"), + extern_func_name="simple_shift", +) +def simple_shift(*args, **kwargs): + ... +``` + +- `library="nvshmem"`:启用 NVSHMEM 设备端 bitcode 链接(`libnvshmem_device.bc`,NVSHMEM 3.7 及以上按 SM 拆分时自动选择对应 `.bc`),并在 kernel 上注册 NVSHMEM cumodule init hook,使设备端 NVSHMEM 调用可以正常初始化。 +- `compiler="clang"`:使用 clang 以 `--cuda-device-only` 编译 `file` 指定的 `.cu` 源文件。 +- `file` / `extern_func_name`:与 CUDA 集成一致,分别指定设备端源文件与其中的函数符号。 + +调用方式与 CUDA 集成相同,通过 `tle_raw.call` 从 Triton kernel 中调用: + +```{code-block} python +@triton.jit +def simple_shift_kernel(destination_ptr): + tle_raw.call(simple_shift, [destination_ptr]) +``` + +NVSHMEM 相关的 host 侧准备工作由 `triton.experimental.tle.raw.nvshmem.utils` 提供: + +- `init_torch_distributed()` / `init_nvshmem_by_torch_pg(common, group)`:基于 PyTorch 进程组初始化 torch.distributed 与 NVSHMEM。 +- `load_host(source)` / `load_common_host(source=None)`:编译并加载 host 侧 `.cu`(含 `nvshmem.h`、`nvshmemx.h`),返回可直接调用其中 `extern "C"` 函数的库对象。 +- `tensor_from_pointer(pointer, shape, dtype, device)`:在 CUDA 显存上创建非拥有的 Torch tensor 视图。 +- `copy_on_stream(dst_ptr, src_ptr, nbytes, stream)`、`set_signal_cuda_ptr(signal_ptr, signal, stream)`:流上的拷贝与 signal 指针设置。 +- `print_perf(...)` / `print_perf_mean(...)`:按 rank 汇总打印性能数据。 +- `enable_nvshmem_device_bc(enabled=True)` / `is_nvshmem_device_bc_enabled()` / `get_nvshmem_extern_libs(arch=None)`:设备端 bitcode 链接开关与外部库查询(`library="nvshmem"` 会自动开启)。 + +使用前需要满足: + +- 设置 `NVSHMEM_HOME` 环境变量(或通过 `triton.knobs.nvidia.nvshmem_home` 指定),用于解析 `libnvshmem_host.so` 与设备端 bitcode。 +- 示例面向 sm90 及以上架构,且要求 `world_size >= 2`;多卡启动时通过 `RANK`、`LOCAL_RANK`、`WORLD_SIZE`、`LOCAL_WORLD_SIZE` 传入拓扑。 + +完整示例见 `python/tutorials/tle/raw/nvshmem/`: + +- `01-simple-shift`:最小的 NVSHMEM 设备端 put 示例,演示 dialect 声明与 host/device 两侧的配合。 +- `02-allgather-gemm`:基于 NVSHMEM 的 all-gather GEMM(含 benchmark)。 +- `03-gemm-allreduce`:基于 NVSHMEM multimem 的 GEMM + all-reduce。 +- `04-cuda-ipc-allreduce`:基于 CUDA IPC 的 all-reduce(含 benchmark)。 + +### 处理流程 + +#### 前端: CUDA-LLVM 集成到 Triton 前端和运行时 + +| 步骤 | 模块 | 关键 Pass 开发 | +|---|---|---| +| 方言注册入口(dialect 装饰器) | `python/triton/experimental/tle/raw/runtime.py` | - 维护 `registry = {"cuda": CUDAJITFunction, "mlir": MLIRJITFunction}`。
- `dialect(name="cuda", ...)` 构造 `CUDAJITFunction` 对象。 | +| TTIR 扩展: `Tle_DSLRegionOp` | `FlagTree/third_party/tle/dialect/include/IR/TleOps.td` | - 接受 Triton 参数;
- 将 LLVM IR 包装到 region 字段中。 | +| CUDA 运行时: 实际调用 clang 的位置 | `python/triton/experimental/tle/raw/cuda/runtime.py` | - `CUDAJITFunction` 在初始化时读取 `.cu` 源文本。
- `make_llvm()` 直接调用 `subprocess.run(clang ...)` 生成 LLVM IR。
- `parse_llvm_ir(...)` 将文本转换为可插入 Triton builder 的模块。
![alt text](../assets/images/make_llvm.PNG) | + + +#### 中端: Python 到 C++,MLIR pass 关系和 pass 继承 + +| 步骤 | 模块 | 关键 Pass 开发 | +|---|---|---| +| 将 LLVM 函数附加到 `dsl_region` | `python/triton/experimental/tle/language/raw/core.py` | - `call()` 获取 builder 上下文。
- 触发 `func.make_llvm(context)`。
- 调用 `create_tle_raw_region_by_llvm_func(...)` 生成 `dsl_region` op。
![alt text](../assets/images/dsl_region_op.PNG) | +| C++ 桥接: IR 注入和类型桥接 | `third_party/tle/triton_tle.cc` | - `third_party/tle/triton_tle.cc` 为 `create_tle_raw_region_by_llvm_func` 和 raw passes 暴露 Python 绑定。
- `third_party/tle/triton_tle_raw.cc` 实现 `createTLERawRegionByLLVMFunc`: 解析函数,将其克隆到当前模块,执行参数/返回类型映射,并创建 `tle::DSLRegionOp` + `tle::YieldOp`。 | + + +#### 后端: CUDA-LLVM IR 转换 — 参数处理和 Triton/TLE-Raw 数据桥接 + +| 步骤 | 模块 | 关键 Pass 开发 | +|---|---|---| +| 后端 pass 注册 | `third_party/nvidia/backend/compiler.py` | - `make_ttgir()` 插入 `tle.raw_passes.add_tle_convert_arg_to_memdesc(pm)`,将 `dsl_region` 张量参数转换为 memdesc 形式。
- `make_llir()` 插入 `tle.raw_passes.add_tle_dsl_region_inline(pm)`,在 LLVM 转换之前将 `dsl_region` 内联到主控制流中。
![alt text](../assets/images/make_ttgir_and_make_llir.png) | +| 参数桥接 | `TleConvertArgToMemDesc`(TTGIR 阶段) | - 将 `dsl_region` 中的张量参数/结果转换为 memdesc 语义,添加本地存储和同步。
- 关键操作:
  - 张量操作数 → `LocalAlloc` + `LocalStore`;
  - `dsl_region` 结果张量 → `LocalLoad` 读回;
  - 必要时插入 `NVVM::Barrier0Op`;
  - 处理 pack 相关的新类型。 | +| LLVM 内联准备 | `TleDSLRegionInline`(LLIR 阶段) | - 从 region op 内联 `tle.dsl_region`。
- 关键操作:
  - 拆分块,创建延续;
  - 将 yield 重写为到延续的分支;
  - 替换原始 `dsl_region` 结果的使用;
  - 删除 `dsl_region` op。 | + + +#### 语义对象映射 + +| 语义对象 | Triton 侧 | TLE-Raw 侧 | LLVM 侧 | +|---|---|---|---| +| 标量参数 | `i32`/`f32`/... | 直接作为 `Tle_ArgType` | LLVM 标量参数 | +| 指针参数 | `tt.ptr` | 直接传递或提取为 LLVM ptr | `attribute((address_space(N))) T*` | +| 张量输入 | `tensor<...>` | 转换为 `ttg.memdesc` / `dsl_region` 操作数 | 展开为 allocated/aligned/offset/sizes/strides | +| 张量输出 | `tensor<...>` | `dsl_region` 结果 + `tle.pack`/yield | LLVM 结构体或多返回字段,然后重新打包 | diff --git a/docs/sglang_plugin_fl_en/dispatch_user_guide/debugg-and-diagonostics.md b/docs/sglang_plugin_fl_en/dispatch_user_guide/debugg-and-diagonostics.md index 9558ec1639..85c68b9f1a 100644 --- a/docs/sglang_plugin_fl_en/dispatch_user_guide/debugg-and-diagonostics.md +++ b/docs/sglang_plugin_fl_en/dispatch_user_guide/debugg-and-diagonostics.md @@ -2,6 +2,23 @@ This section introduces diagnostics on ops dispatch. + +## Policy and backend resolution + +Dispatch is policy-based rather than a fixed `flagos > vendor > reference` chain. When an operation does not use the expected implementation, check the effective configuration and backend availability: + +1. Check environment-variable overrides such as `SGLANG_FL_PREFER`, `SGLANG_FL_PER_OP`, `SGLANG_FL_ALLOW_VENDORS`, `SGLANG_FL_DENY_VENDORS`, and `SGLANG_FL_STRICT`. +2. Check `SGLANG_FL_CONFIG` and the detected platform YAML under `sglang_fl/dispatch/config/`. +3. Enable `SGLANG_FL_DISPATCH_DEBUG=[TODO: needs confirmation]` if additional policy diagnostics are required. +4. Confirm that the selected backend's runtime and device are available in the target image. + +Strict mode reports an error instead of falling back when the preferred or explicitly ordered backend is unavailable. Without strict mode, dispatch filters unavailable or disallowed candidates and selects the next allowed implementation. + +## Distributed and FlagCX diagnostics + +For distributed communication failures, verify the selected `SGLANG_FL_DIST_BACKEND`, `FLAGCX_PATH` when FlagCX is expected, device visibility, network configuration, and tensor/pipeline parallel settings. Platform-specific framework, image, and network-device prerequisites are documented on the [centralized vendor/framework/image-selection page](https://flagos.io/resourcedownload?lang=en). + + ## Dispatch log See which backend each fused op resolved to (written at server startup): diff --git a/docs/sglang_plugin_fl_en/dispatch_user_guide/dispatch-through-environment-variables.md b/docs/sglang_plugin_fl_en/dispatch_user_guide/dispatch-through-environment-variables.md index bb0cedf381..d1eb57faac 100644 --- a/docs/sglang_plugin_fl_en/dispatch_user_guide/dispatch-through-environment-variables.md +++ b/docs/sglang_plugin_fl_en/dispatch_user_guide/dispatch-through-environment-variables.md @@ -1,7 +1,16 @@ # Dispatch through environment variables -All plugin behavior is controlled by environment variables with the `SGLANG_FL_*` prefix. +All plugin behavior is controlled by environment variables with the `SGLANG_FL_*` prefix. + +The effective configuration precedence is: + +```{code-block} text +environment variables > explicit YAML (`SGLANG_FL_CONFIG`) > platform auto-detected YAML > code defaults +``` + +Environment variables can override explicit and platform YAML values. Platform-specific backend names and runtime defaults depend on the target platform; use `[TODO: needs confirmation]` where a deployment requires an accepted-value list that is not defined by this generic guide. + ## Layer 2 — Fused Op Dispatch @@ -12,10 +21,11 @@ All plugin behavior is controlled by environment variables with the `SGLANG_FL_ | `SGLANG_FL_PER_OP` | — | Per-op backend priority, e.g. `rms_norm=vendor\|flagos;silu_and_mul=reference` | | `SGLANG_FL_OOT_BLACKLIST` | — | Skip listed ops from OOT dispatch (comma-separated class names) | | `SGLANG_FL_OOT_WHITELIST` | — | Only dispatch listed ops (mutually exclusive with BLACKLIST) | -| `SGLANG_FL_STRICT` | `0` | `1` = disable fallback (error if preferred backend unavailable) | -| `SGLANG_FL_DENY_VENDORS` | — | Deny specific vendors (comma-separated, e.g. `cuda,ascend`) | +| `SGLANG_FL_STRICT` | `0` | `1` disables fallback; an unavailable preferred/ordered backend becomes an error | +| `SGLANG_FL_DENY_VENDORS` | — | Deny specific vendors (comma-separated) | | `SGLANG_FL_ALLOW_VENDORS` | — | Allow only listed vendors (comma-separated) | | `SGLANG_FL_DISPATCH_LOG` | — | Path to dispatch log file (records which ops are intercepted) | +| `SGLANG_FL_DISPATCH_DEBUG` | [TODO: needs confirmation] | Enables additional dispatch diagnostics; accepted values are [TODO: needs confirmation] | ## Layer 1 — ATen Replacement (FlagGems) @@ -34,15 +44,17 @@ All plugin behavior is controlled by environment variables with the `SGLANG_FL_ | Variable | Default | Description | |----------|---------|-------------| -| `SGLANG_FL_DIST_BACKEND` | `nccl` | Backend: `nccl` / `hccl` / `flagcx` | -| `FLAGCX_PATH` | — | FlagCX installation path (if set, defaults to `flagcx` backend) | +| `SGLANG_FL_DIST_BACKEND` | `nccl` in the generic reference table | Distributed backend selection; platform runtime defaults may map to another backend, and `FLAGCX_PATH` can select FlagCX when no explicit override is set | +| `FLAGCX_PATH` | — | FlagCX installation path; required for FlagCX collectives | + +Supported runtime choices depend on the platform and installed libraries. Use `[TODO: needs confirmation]` for platform-specific accepted values not covered by the target runtime. ## System / Debug | Variable | Default | Description | |----------|---------|-------------| | `SGLANG_FL_CONFIG` | — | Path to YAML config file (overrides platform auto-detection) | -| `SGLANG_FL_PLATFORM` | (auto) | Force platform: `cuda`, `ascend` (overrides auto-detection) | +| `SGLANG_FL_PLATFORM` | (auto) | Force platform; accepted values are [TODO: needs confirmation] | | `SGLANG_FL_LOG_LEVEL` | `INFO` | Dispatch system log level: `DEBUG`, `INFO`, `WARNING`, `ERROR` | | `SGLANG_PLUGINS` | (all) | SGLang built-in: filter which plugins to load (comma-separated) | diff --git a/docs/sglang_plugin_fl_en/dispatch_user_guide/dispatch-through-yaml-file.md b/docs/sglang_plugin_fl_en/dispatch_user_guide/dispatch-through-yaml-file.md index 97d0133025..7defefed81 100644 --- a/docs/sglang_plugin_fl_en/dispatch_user_guide/dispatch-through-yaml-file.md +++ b/docs/sglang_plugin_fl_en/dispatch_user_guide/dispatch-through-yaml-file.md @@ -1,18 +1,23 @@ # Dispatch through YAML config file -The plugin ships with a sample config file `config/sample.yaml` with all available options. Copy it and customize: + +The plugin ships platform YAML defaults under `sglang_fl/dispatch/config/`. Available platform files include `ascend.yaml`, `gcu.yaml`, `hygon.yaml`, `iluvatar.yaml`, `kunlunxin.yaml`, `musa.yaml`, `nvidia.yaml`, and `tsingmicro.yaml`. These files provide default dispatch policy; they are not a vendor validation matrix. -```{code-block} shell -# Copy the sample config -cp $(python -c "from sglang_fl.config import _CONFIG_DIR; print(_CONFIG_DIR / 'sample.yaml')") my_config.yaml +Create an explicit YAML file when you want to override the platform defaults: -# Edit as needed, then launch with it +```{code-block} shell SGLANG_FL_CONFIG=./my_config.yaml python -m sglang.launch_server \ --model-path Qwen/Qwen2.5-0.5B-Instruct \ --port 30000 --disable-piecewise-cuda-graph ``` -If `SGLANG_FL_CONFIG` is not set, the plugin uses sensible defaults (equivalent to `prefer: flagos` on CUDA). You only need a YAML file when you want to customize behavior. +Configuration precedence is: + +```{code-block} text +environment variables > explicit YAML (`SGLANG_FL_CONFIG`) > platform auto-detected YAML > code defaults +``` + +If no explicit YAML is supplied, the detected platform YAML and code defaults are used. Environment variables can override either YAML source. ## Config Fields @@ -25,8 +30,7 @@ op_backends: rms_norm: [vendor, flagos, reference] silu_and_mul: [flagos, vendor, reference] -# Layer 2 fused ops to skip (fall through to SGLang native CUDA) -# Available: SiluAndMul, RMSNorm, RotaryEmbedding +# Layer 2 fused ops to skip (fall through to SGLang native path) oot_blacklist: - RotaryEmbedding @@ -34,20 +38,30 @@ oot_blacklist: flagos_blacklist: - mul - sub + +# Optional vendor filters +allow_vendors: [] +deny_vendors: [] + +# Disable fallback when true +strict: false ``` | Field | Description | |-------|-------------| | `prefer` | Global backend preference: `flagos`, `vendor`, `reference` | -| `op_backends` | Per-op ordered backend list (first available wins, can list 1–3 backends) | -| `oot_blacklist` | Layer 2 fused ops to skip from OOT dispatch (fall through to SGLang native CUDA) | -| `flagos_blacklist` | Layer 1 ATen ops to exclude from FlagGems replacement (fall through to PyTorch native) | +| `op_backends` | Per-op ordered backend list; the first available and allowed backend is selected | +| `oot_blacklist` | Layer 2 fused ops to skip from OOT dispatch | +| `flagos_blacklist` | Layer 1 ATen ops to exclude from FlagGems replacement | +| `allow_vendors` | Optional vendor allow list; only listed vendors are eligible | +| `deny_vendors` | Optional vendor deny list | +| `strict` | Disables fallback when enabled; exact YAML boolean syntax is [TODO: needs confirmation] | ## Common Recipes -Each recipe shows a YAML config and expected dispatch result. Use [Dispatch Log](/dispatch_user_guide/debugg-and-diagonostics.md) to verify. +Each recipe shows a YAML config and expected dispatch result. Use [Dispatch Log](debugg-and-diagonostics.md) to verify. -### 1. Skip RotaryEmbedding from OOT dispatch (fall through to SGLang native CUDA) +### 1. Skip RotaryEmbedding from OOT dispatch (fall through to SGLang native path) ```yaml # my_config.yaml @@ -67,7 +81,7 @@ op_backends: rms_norm: [vendor, flagos, reference] ``` -Expected dispatch log: `RMSNorm → vendor(vendor.nvidia)`, `SiluAndMul → flagos(flagos)`. +Expected dispatch log: the first available, allowed backend is selected according to the active platform and vendor filters. ### 3. Use pure PyTorch reference for all Ops (useful for precision debugging) @@ -76,4 +90,4 @@ Expected dispatch log: `RMSNorm → vendor(vendor.nvidia)`, `SiluAndMul → flag prefer: reference ``` -Expected dispatch log: all ops → `reference(reference)`. +Expected dispatch log: reference implementations are selected when available. diff --git a/docs/sglang_plugin_fl_en/dispatch_user_guide/dispatch-user-guide.md b/docs/sglang_plugin_fl_en/dispatch_user_guide/dispatch-user-guide.md index 19a919600f..5b9f28bd84 100644 --- a/docs/sglang_plugin_fl_en/dispatch_user_guide/dispatch-user-guide.md +++ b/docs/sglang_plugin_fl_en/dispatch_user_guide/dispatch-user-guide.md @@ -1,6 +1,7 @@ # Operator Dispatch User Guide -The dispatch system provides three layers of operator replacement. You can control each layer independently and flexibly. + +The dispatch system provides operator replacement and distributed communication configuration through YAML files and environment variables. You can control the replacement layers independently and configure platform-aware communication backends. The dispatch system supports both YAML configuration and environment variables for fine-grained control. Environment variables take precedence over YAML config. @@ -10,8 +11,6 @@ The priority chain is as follows: SGLANG_FL_* env vars > YAML config (SGLANG_FL_CONFIG) > Platform auto-detect YAML > Code defaults ``` - - ```{toctree} :maxdepth: 2 diff --git a/docs/sglang_plugin_fl_en/dispatch_user_guide/vendor-integration.md b/docs/sglang_plugin_fl_en/dispatch_user_guide/vendor-integration.md index 87f110614c..47ab7465cc 100644 --- a/docs/sglang_plugin_fl_en/dispatch_user_guide/vendor-integration.md +++ b/docs/sglang_plugin_fl_en/dispatch_user_guide/vendor-integration.md @@ -7,15 +7,10 @@ cp -r sglang_fl/dispatch/backends/vendor/template/ \ sglang_fl/dispatch/backends/vendor/my_chip/ ``` -The following table lists the existing vendors: + +Vendor-specific framework selection, runtime images, validation status, installation commands, and adaptation procedures are maintained on the centralized page: [centralized vendor/framework/image-selection page](https://flagos.io/resourcedownload?lang=en). The generic contract below does not imply that every backend supports every operation or model. -| Vendor | Directory | Hardware Detection | -| :--- | :--- | :--- | -| NVIDIA CUDA | `vendor/cuda/` | `sgl_kernel` importable | -| Huawei Ascend | `vendor/ascend/` | `torch_npu` importable | -| Template | `vendor/template/` | Always False (reference only) | - -To integrate with a new vendor, you need to implement two files: +To integrate with a new vendor, implement the backend contract and registration described below. The three operations in the example are a minimal illustration, not the complete v0.2.0 operation surface. ## 1. Backend class (my_chip.py) @@ -115,9 +110,18 @@ Each op function receives standardized arguments (same as vllm-plugin-FL): | `rms_norm` | `fn(obj, x: Tensor, residual: Optional[Tensor] = None) -> Tensor \| tuple[Tensor, Tensor]` | | `rotary_embedding` | `fn(obj, query, key, cos, sin, position_ids, rotary_interleaved=False, inplace=True) -> tuple[Tensor, Tensor]` | - The `obj` parameter provides access to layer attributes (`obj.weight`, `obj.variance_epsilon`, etc.). These attribute names are identical between SGLang and vLLM, so the same impl works for both frameworks. -## Vendor's backend auto-discovery +## Vendor backend auto-discovery + +The plugin scans `dispatch/backends/vendor/*/register_ops.py` at startup. If `is_available()` returns True, the vendor's ops are registered. No other files need modification. + + +Optional integration hooks can add platform-specific behavior when needed: + +- `patch.py` can apply framework patches required to expose a backend or transfer path. +- `register_platform.py` can register platform identity and runtime behavior. +- A platform YAML under `sglang_fl/dispatch/config/` can provide default dispatch policy for the detected platform. -The plugin scans `dispatch/backends/vendor/*/register_ops.py` at startup. If is_available() returns True, the vendor's ops are registered. No other files need modification. \ No newline at end of file +These files extend the generic integration contract; they do not establish a universal vendor support or validation claim. Detailed adaptation procedures remain on the centralized page: [centralized vendor/framework/image-selection page](https://flagos.io/resourcedownload?lang=en). + diff --git a/docs/sglang_plugin_fl_en/getting_started/getting-started.md b/docs/sglang_plugin_fl_en/getting_started/getting-started.md index 708022e2cf..05877105bc 100644 --- a/docs/sglang_plugin_fl_en/getting_started/getting-started.md +++ b/docs/sglang_plugin_fl_en/getting_started/getting-started.md @@ -7,6 +7,7 @@ This section covers the requirements for installing sglang-plugin-FL and guides requirements.md install.md -run-inference-task.md + +quick-run-inference-task.md ``` diff --git a/docs/sglang_plugin_fl_en/getting_started/install.md b/docs/sglang_plugin_fl_en/getting_started/install.md index c3c829a72c..feed108f4c 100644 --- a/docs/sglang_plugin_fl_en/getting_started/install.md +++ b/docs/sglang_plugin_fl_en/getting_started/install.md @@ -1,28 +1,25 @@ # Install sglang-plugin-FL -## Docker Images (Recommended) - -Pre-built Docker images for v0.1.0-rc2: - -| Platform | Image | Contents | -|----------|-------|----------| -| NVIDIA GPU (dual-node) | `harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-nvidia-dual` | sglang 0.5.11, flag_gems 5.3.0rc2, torch 2.11.0, triton 3.6.0 | -| NVIDIA GPU (single-node) | `harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-nvidia-single` | sglang 0.5.11, flag_gems 5.3.0rc2, torch 2.11.0, triton 3.6.0 | -| Moore Threads MUSA | `harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-musa` | sglang 0.5.12, torch 2.9.0, flag_gems 5.0.2 | -| Moore Threads (SVT) | `harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-mthreads-svt` | sglang 0.5.11, flag_gems 5.3.0rc2, torch 2.9.0, triton 3.1.0 | -| Huawei Ascend | `harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-ascend` | sglang 0.5.12, flag_gems 5.0.2, CANN 8.5.0 | - -```bash -# NVIDIA dual-node (cross-node inference) -docker pull harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-nvidia-dual -# NVIDIA single-node -docker pull harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-nvidia-single -# Moore Threads MUSA -docker pull harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-musa -# Moore Threads SVT (full-stack test) -docker pull harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-mthreads-svt -# Huawei Ascend -docker pull harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-ascend -``` - -Dual-node images support cross-node LLM inference. Single-node images support single-machine inference. SVT images are full-stack test images. + +## Runtime images and platform packages + +Vendor-specific framework selection, runtime images, validation status, installation commands, and adaptation procedures are maintained on the centralized page: [centralized vendor/framework/image-selection page](https://flagos.io/resourcedownload?lang=en). Use that page to choose the vendor, framework, and image before installing or launching sglang-plugin-FL. + +This page intentionally does not maintain a full image matrix or platform-specific package recipe. + + +## Empty mode + +Empty mode is an installation/runtime assembly mechanism for platforms where the CUDA-oriented SGLang dependency stack is not the target deployment environment. It avoids treating CUDA packages as the universal dependency set, but it is **not** a no-device mode. + +The target platform still provides: + +- vendor torch; +- drivers and firmware; +- device runtime; +- communication libraries; +- platform attention backends; +- operators not covered by sglang-plugin-FL. + +Use the centralized vendor/framework/image-selection page for the platform-specific image and dependency set: [centralized vendor/framework/image-selection page](https://flagos.io/resourcedownload?lang=en). + diff --git a/docs/sglang_plugin_fl_en/getting_started/quick-run-inference-task.md b/docs/sglang_plugin_fl_en/getting_started/quick-run-inference-task.md index afa9e4bbd2..73454e1b88 100644 --- a/docs/sglang_plugin_fl_en/getting_started/quick-run-inference-task.md +++ b/docs/sglang_plugin_fl_en/getting_started/quick-run-inference-task.md @@ -1,6 +1,6 @@ # Quick run an inference task -This section covers how to quick start a inference task through sglang-plugin-FL. +This section covers how to quick start a inference task through sglang-plugin-FL. ## Installation @@ -32,6 +32,10 @@ cd FlagCX && make USE_NVIDIA=1 export FLAGCX_PATH="$PWD" ``` + +For vendor-specific Empty-mode installation, runtime images, framework packages, and validation prerequisites, use the centralized vendor/framework/image-selection page: [centralized vendor/framework/image-selection page](https://flagos.io/resourcedownload?lang=en). Empty mode is not a no-device mode; the target platform must provide its vendor torch, drivers, firmware, device runtime, communication libraries, attention backend, and uncovered operators. + + ## Download models ```{code-block} shell @@ -76,6 +80,22 @@ python -m sglang.launch_server \ FlagGems Triton kernels contain `logging.Logger` calls that are incompatible with `torch.compile` (used by SGLang's piecewise CUDA graph). Always use `--disable-piecewise-cuda-graph` when launching the server. Regular CUDA graph capture works normally. ``` + +### Multinode and pipeline parallelism + +For multinode inference, configure the distributed backend and network interfaces for the target platform, then use the corresponding SGLang tensor-parallel and pipeline-parallel arguments. The repository's multinode examples are starting points rather than universal hardware recipes; replace addresses, device visibility variables, communication paths, and interface names with values for the target deployment. + +Pipeline-parallel workflows may use FlagCX or `torch.distributed` through `CommunicatorFL`. See the dispatch environment-variable guide for backend selection. Platform-specific images, framework builds, and validation status are maintained on the centralized page: [centralized vendor/framework/image-selection page](https://flagos.io/resourcedownload?lang=en). + +### Qwen3.6 MTP + +The v0.2.0 examples include Qwen3.6 multi-token prediction (MTP) workflow coverage. Use the model's supported SGLang launch arguments and the target platform's validated runtime selection; this page does not assert universal model or hardware support. + +### Throughput benchmarking + +Use the repository's serving benchmark tooling after the inference path is working. Compare throughput, time to first token, and decode latency across input/output lengths and request rates. Do not infer universal performance characteristics from the generic workflow. + + ### 2. Send a Request After the server is ready (`The server is fired up and ready to roll`), send a request: diff --git a/docs/sglang_plugin_fl_en/getting_started/requirements.md b/docs/sglang_plugin_fl_en/getting_started/requirements.md index 931b80ded9..5887451435 100644 --- a/docs/sglang_plugin_fl_en/getting_started/requirements.md +++ b/docs/sglang_plugin_fl_en/getting_started/requirements.md @@ -2,29 +2,32 @@ ## Software requirements -The following software versions are required for sglang-plugin-FL. + +The required runtime stack depends on the selected vendor, framework, and image. Use the centralized vendor/framework/image-selection page for platform-specific SDKs, framework builds, images, validation status, and package versions: [centralized vendor/framework/image-selection page](https://flagos.io/resourcedownload?lang=en). -| Package | Version | +Generic requirements for sglang-plugin-FL are: + +| Component | Requirement | |---------|---------| -| SGLang | 0.5.11 | -| sglang-kernel | 0.4.2 | -| PyTorch | 2.11.0+cu130 | -| Triton | 3.6.0 | -| FlagGems | 4.2.1rc0 | -| flashinfer | 0.6.8.post1 | -| Python | 3.12 | -| CUDA | 13.0 | +| SGLang | Compatible SGLang runtime with plugin entry-point loading | +| sglang-plugin-FL | Installed in the target runtime environment | +| FlagGems | Required for Layer 1 ATen replacement and FlagGems-backed fused implementations when enabled | +| FlagCX | Required when using FlagCX collectives | +| Vendor runtime stack | Vendor torch, drivers, firmware, device runtime, communication libraries, attention backend, and uncovered operators | + + +## Empty mode boundary + +Empty mode is not a no-device mode. It changes installation/runtime assembly so that vendor-specific environments can provide their own runtime stack instead of inheriting a CUDA-oriented dependency set. The target platform still supplies vendor torch, drivers, firmware, device runtime, communication libraries, platform attention backend, and operators not covered by the plugin. + ## Hardware requirements -- NVIDIA GPU with CUDA 13.0 support, or +- NVIDIA GPU with CUDA support, or - Huawei Ascend NPU with CANN toolkit, or - Other supported hardware with appropriate vendor SDK ## Verified models -| Model | TP | Status | -|-------|-----|--------| -| Qwen3.6-27B (Hybrid Attention + FLA + MoE) | tp=1 | Verified | -| Qwen3.6-35B-A3B (MoE, 256 experts) | tp=1 | Verified | -| Qwen2.5-14B-Instruct | tp=8 | Verified | + +Detailed model, quantization, platform, and validation status is maintained on the centralized page: [centralized vendor/framework/image-selection page](https://flagos.io/resourcedownload?lang=en). Do not treat this page as a complete vendor support matrix. diff --git a/docs/sglang_plugin_fl_en/index.md b/docs/sglang_plugin_fl_en/index.md index 04aa8a8b1a..998c58022c 100644 --- a/docs/sglang_plugin_fl_en/index.md +++ b/docs/sglang_plugin_fl_en/index.md @@ -42,13 +42,13 @@ Guides you how to dispatch operators between FlagGems, vendor-specific, and PyTo ::: :::{grid-item-card} {octicon}`code;1.5em;sd-mr-1` API Reference -:link: reference/dispatch-api-reference +:link: reference/dispatch-reference-and-example :link-type: doc API reference for the operator dispatch system and configuration options. +++ -[Learn more »](reference/dispatch-api-reference.md) +[Learn more »](reference/dispatch-reference-and-example.md) ::: :::: @@ -78,5 +78,5 @@ dispatch_user_guide/dispatch-user-guide.md :maxdepth: 5 :hidden: -reference/dispatch-api-reference.md +reference/dispatch-reference-and-example.md ``` diff --git a/docs/sglang_plugin_fl_en/overview/features.md b/docs/sglang_plugin_fl_en/overview/features.md index 2490745263..ee049352f1 100644 --- a/docs/sglang_plugin_fl_en/overview/features.md +++ b/docs/sglang_plugin_fl_en/overview/features.md @@ -2,7 +2,8 @@ SGLang's inference engine relies on NVIDIA-specific components: flashinfer for attention, sgl_kernel for fused CUDA kernels, and NCCL for distributed communication. Running on alternative hardware (Huawei Ascend, Cambricon MLU, Iluvatar, etc.) would otherwise require invasive source modifications. -This plugin provides a non-intrusive adaptation layer through three levels of replacement: + +This plugin provides a non-intrusive adaptation layer through three levels of replacement and extension: ## Layer 1 — ATen Operators @@ -10,15 +11,18 @@ Replaces PyTorch's low-level ops (matmul, softmax, embedding, etc.) with FlagGem ## Layer 2 — SGLang Fused Kernels -Intercepts SGLang's custom fused ops (SiluAndMul, RMSNorm, RotaryEmbedding) via HookRegistry AROUND hooks, routing through a standardized dispatch system (aligned with vllm-plugin-FL) to select the best available backend: +Intercepts SGLang's custom fused ops via HookRegistry AROUND hooks, routing through a standardized dispatch system (aligned with vllm-plugin-FL) to select the best available backend: -- **FlagGems** — Triton-based implementations (default, highest priority) -- **Vendor** — Chip-native implementations (e.g., CUDA sgl_kernel, Ascend CANN) +- **FlagGems** — Triton-based implementations +- **Vendor** — Chip-native implementations - **Reference** — Pure PyTorch fallback implementations +The dispatch policy combines global backend preference, per-operation backend ordering, backend availability, vendor allow/deny filters, strict mode, and fallback behavior. Do not interpret the list above as a fixed priority order for every operation. + ## Layer 3 — Distributed Communication -Replaces NCCL-based collectives with CommunicatorFL (backed by FlagCX or torch.distributed), enabling multi-card inference on any hardware. Supports all_reduce, all_gather, reduce_scatter, send, and recv operations. +Replaces NCCL-based collectives with CommunicatorFL (backed by FlagCX or torch.distributed), enabling multi-card inference on different hardware backends. Supports all_reduce, all_gather, reduce_scatter, send, and recv operations, with platform-aware backend selection and pipeline-parallel communication support. + ![alt text](../assets/sglang-plugin-fl-arch.png) diff --git a/docs/sglang_plugin_fl_en/overview/how-plugin-works.md b/docs/sglang_plugin_fl_en/overview/how-plugin-works.md index 5ff0200093..aad982f75d 100644 --- a/docs/sglang_plugin_fl_en/overview/how-plugin-works.md +++ b/docs/sglang_plugin_fl_en/overview/how-plugin-works.md @@ -18,6 +18,7 @@ sglang_fl = "sglang_fl:activate_platform" The core mechanism uses an AROUND hook on `MultiPlatformOp.dispatch_forward()` combined with a standardized dispatch system: + ```{code-block} python dispatch_forward() called for an op (e.g. RMSNorm) → AROUND hook intercepts @@ -28,11 +29,17 @@ dispatch_forward() called for an op (e.g. RMSNorm) rms_norm_bridge(self, x, residual, post_residual_addition) → Bridge handles SGLang-specific params (post_residual_addition → merge into residual) → Bridge calls dispatch.call_op("rms_norm", obj, x, residual) - → OpManager resolves best impl via policy (flagos > vendor > reference) - → Calls the selected backend: rms_norm_flaggems(obj, x, residual) + → OpManager asks SelectionPolicy for candidate backend order + → Policy combines global preference, per-op ordering, availability, vendor filters, and strict mode + → OpManager selects the first available implementation, caches the result, or falls back when allowed + → Calls the selected backend implementation ``` The bridge layer decouples framework-specific parameters from the standardized op signatures. Vendor backends only need to implement the standard signatures — the same impl works for both sglang-plugin-FL and vllm-plugin-FL. + +Strict mode disables fallback: if the preferred or explicitly ordered backend cannot be used, dispatch reports an error instead of silently selecting another backend. Without strict mode, unavailable backends are filtered out and the next allowed candidate can run. + + ## Dispatch Architecture (shared with vllm-plugin-FL) ```{code-block} python @@ -50,11 +57,9 @@ The bridge layer decouples framework-specific parameters from the standardized o ┌────────────────┼────────────────┐ ▼ ▼ ▼ ┌─────────────┐ ┌───────────┐ ┌──────────────┐ - │ DEFAULT │ │ VENDOR │ │ REFERENCE │ - │ (FlagGems) │ │ (Ascend/ │ │ (PyTorch) │ - │ priority=150│ │ CUDA) │ │ priority=50 │ - │ │ │ priority= │ │ │ - │ │ │ 100 │ │ │ + │ FLAGOS │ │ VENDOR │ │ REFERENCE │ + │ (FlagGems) │ │ (chip- │ │ (PyTorch) │ + │ │ │ native) │ │ │ └─────────────┘ └───────────┘ └──────────────┘ ``` Chip vendors implement the **same backend interface** for both frameworks. The only framework-specific code is the bridge layer, which is maintained by the plugin. @@ -68,3 +73,11 @@ Plugin loads → flag_gems.enable(record=True) → _AtenOnlyFilter ensures only flag_gems.ops.* calls are recorded (excludes internal FlagGems calls from Layer 2 flagos implementations) ``` + + +## Empty mode boundary + +Empty mode is an installation/runtime assembly mechanism for target platforms where the CUDA-oriented dependency set is not the deployment environment. It is not a no-device mode. The target platform still supplies vendor torch, drivers, firmware, device runtime, communication libraries, platform attention backends, and any operators not covered by the plugin. + +Use the centralized vendor/framework/image-selection page for platform-specific runtime and image choices: [centralized vendor/framework/image-selection page](https://flagos.io/resourcedownload?lang=en). + diff --git a/docs/sglang_plugin_fl_en/overview/project-structure.md b/docs/sglang_plugin_fl_en/overview/project-structure.md index 42780f1518..5fab368b4f 100644 --- a/docs/sglang_plugin_fl_en/overview/project-structure.md +++ b/docs/sglang_plugin_fl_en/overview/project-structure.md @@ -1,5 +1,6 @@ # Project structure + ```{code-block} python sglang_fl/ ├── pyproject.toml # Package config + entry_points registration @@ -11,11 +12,6 @@ sglang_fl/ │ ├── communicator.py # CommunicatorFL (FlagCX / torch.distributed wrapper) │ └── device_communicators/ │ └── flagcx.py # FlagCX-specific communicator - ├── config/ - │ ├── __init__.py # YAML config loader with platform auto-detection - │ ├── sample.yaml # Full example config with all options documented - │ ├── nvidia.yaml # NVIDIA CUDA platform defaults - │ └── ascend.yaml # Ascend platform defaults (with blacklists) └── dispatch/ # Op dispatch system (aligned with vllm-plugin-FL) ├── __init__.py # Public API: call_op(), resolve_op() ├── types.py # OpImpl, BackendImplKind, BackendPriority @@ -25,6 +21,15 @@ sglang_fl/ ├── builtin_ops.py # Registration orchestrator ├── ops.py # FLBackendBase ABC (op signature definitions) ├── logger_manager.py # Logging with SGLANG_FL_LOG_LEVEL + ├── config/ # Platform YAML defaults and dispatch configuration + │ ├── ascend.yaml + │ ├── gcu.yaml + │ ├── hygon.yaml + │ ├── iluvatar.yaml + │ ├── kunlunxin.yaml + │ ├── musa.yaml + │ ├── nvidia.yaml + │ └── tsingmicro.yaml ├── bridge/ # SGLang ↔ dispatch parameter translation │ ├── __init__.py │ ├── silu_and_mul.py # forward_cuda(self, x) → call_op("silu_and_mul", obj, x) @@ -41,8 +46,10 @@ sglang_fl/ │ ├── register_ops.py │ └── impl/ # activation.py, normalization.py, rotary.py └── vendor/ # VENDOR backends (auto-discovered) - ├── ascend/ # Huawei Ascend NPU (torch_npu) - ├── cuda/ # NVIDIA CUDA (sgl_kernel) + ├── / # Vendor-specific backend package └── template/ # Template for new vendors ``` + +Platform YAML files provide default dispatch policy for a detected platform. They are not a vendor validation matrix. Vendor-specific framework selection, runtime images, validation status, installation commands, and adaptation procedures are maintained on the centralized page: [centralized vendor/framework/image-selection page](https://flagos.io/resourcedownload?lang=en). + diff --git a/docs/sglang_plugin_fl_en/reference/dispatch-reference-and-example.md b/docs/sglang_plugin_fl_en/reference/dispatch-reference-and-example.md index 710801e7b0..80316d6699 100644 --- a/docs/sglang_plugin_fl_en/reference/dispatch-reference-and-example.md +++ b/docs/sglang_plugin_fl_en/reference/dispatch-reference-and-example.md @@ -2,6 +2,16 @@ This page documents the references and examples of environment variables for sglang-plugin-FL. + +The effective configuration precedence is: + +```{code-block} text +environment variables > explicit YAML (`SGLANG_FL_CONFIG`) > platform auto-detected YAML > code defaults +``` + +For vendor/framework/image selection, platform-specific runtime packages, images, and validation status, use the [centralized vendor/framework/image-selection page](https://flagos.io/resourcedownload?lang=en). + + ## Environment Variables — Complete Reference ### Layer 2 — Fused Op Dispatch diff --git a/docs/sglang_plugin_fl_en/release_notes/release-notes.md b/docs/sglang_plugin_fl_en/release_notes/release-notes.md index af86697a1d..9eea075c2a 100644 --- a/docs/sglang_plugin_fl_en/release_notes/release-notes.md +++ b/docs/sglang_plugin_fl_en/release_notes/release-notes.md @@ -1,5 +1,22 @@ # Release Notes + +## v0.2.0 + +This release expands sglang-plugin-FL's dispatch, platform configuration, and runtime configuration surface for `v0.1.0 → v0.2.0`. + +- Added features + - Expanded platform configuration through platform YAML defaults under `sglang_fl/dispatch/config/`. + - Expanded dispatch policy controls, including global backend preference, per-operation backend ordering, vendor allow/deny filtering, strict mode, fallback behavior, and dispatch cache invalidation. + - Empty-mode installation/runtime assembly boundary for vendor-specific runtime stacks. Empty mode is not a no-device mode; the target platform still provides vendor torch, drivers, firmware, device runtime, communication libraries, and uncovered operators. + - Multinode and pipeline-parallel inference examples, Qwen3.6 MTP workflow coverage, and throughput benchmark infrastructure. + +- Documentation scope + + - Vendor-specific framework selection, runtime images, validation status, installation commands, and adaptation procedures are centralized outside this English plugin documentation: [centralized vendor/framework/image-selection page](https://flagos.io/resourcedownload?lang=en). + - This documentation keeps the generic plugin architecture, dispatch configuration, and workflow shape without maintaining a full vendor support or image matrix. + + ## v0.1.0 @@ -19,4 +36,4 @@ Initial release of sglang-plugin-FL. - Support for NVIDIA CUDA, Huawei Ascend, and extensible to other hardware - Verified models: Qwen3.6-27B, Qwen3.6-35B-A3B, Qwen2.5-14B-Instruct - Dispatch logging and ATen replacement logging for debugging - - Precision bisection workflow for numerical debugging \ No newline at end of file + - Precision bisection workflow for numerical debugging diff --git a/docs/sglang_plugin_fl_zh/dispatch_user_guide/debugg-and-diagonostics.md b/docs/sglang_plugin_fl_zh/dispatch_user_guide/debugg-and-diagonostics.md index d69578f11d..723fae0438 100644 --- a/docs/sglang_plugin_fl_zh/dispatch_user_guide/debugg-and-diagonostics.md +++ b/docs/sglang_plugin_fl_zh/dispatch_user_guide/debugg-and-diagonostics.md @@ -2,6 +2,23 @@ 本节介绍算子调度的诊断方法。 + +## 策略与后端解析 + +调度采用基于策略的方式,而不是固定的 `flagos > vendor > reference` 链。当算子未使用预期实现时,请检查有效配置和后端可用性: + +1. 检查 `SGLANG_FL_PREFER`、`SGLANG_FL_PER_OP`、`SGLANG_FL_ALLOW_VENDORS`、`SGLANG_FL_DENY_VENDORS` 和 `SGLANG_FL_STRICT` 等环境变量覆盖项。 +2. 检查 `SGLANG_FL_CONFIG` 以及 `sglang_fl/dispatch/config/` 下检测到的平台 YAML。 +3. 如果需要额外的策略诊断,请启用 `SGLANG_FL_DISPATCH_DEBUG=[TODO: needs confirmation]`。 +4. 确认所选后端的运行时和设备在目标镜像中可用。 + +严格模式下,如果首选或显式排序的后端不可用,系统会报错而不是回退。未启用严格模式时,调度会过滤不可用或不允许的候选项,并选择下一个允许的实现。 + +## 分布式与 FlagCX 诊断 + +对于分布式通信故障,请检查所选的 `SGLANG_FL_DIST_BACKEND`、预期使用 FlagCX 时的 `FLAGCX_PATH`、设备可见性、网络配置以及张量并行 / 流水线并行设置。平台特定的框架、镜像和网络设备前置条件记录在[集中式厂商 / 框架 / 镜像选择页面](https://flagos.io/resourcedownload?lang=en)中。 + + ## 调度日志 查看每个融合算子解析到哪个后端(在服务器启动时写入): @@ -40,7 +57,7 @@ sort -u /tmp/gems_aten.txt ## 通过精度二分法排查数值精度问题 -当出现数值差异时,隔离出导致问题的层。如果输出在第 N 步发散但第 N-1 步正常,则问题层/算子被定位。 +当出现数值差异时,隔离出导致问题的层。如果输出在第 N 步发散但第 N-1 步正常,则问题层 / 算子被定位。 ```{code-block} python # 第 1 步:禁用所有——确认原生 SGLang 正常工作 diff --git a/docs/sglang_plugin_fl_zh/dispatch_user_guide/dispatch-through-environment-variables.md b/docs/sglang_plugin_fl_zh/dispatch_user_guide/dispatch-through-environment-variables.md index 0881aa824d..387e3f8bc6 100644 --- a/docs/sglang_plugin_fl_zh/dispatch_user_guide/dispatch-through-environment-variables.md +++ b/docs/sglang_plugin_fl_zh/dispatch_user_guide/dispatch-through-environment-variables.md @@ -1,21 +1,31 @@ # 通过环境变量进行调度 -所有插件行为由带有 `SGLANG_FL_*` 前缀的环境变量控制。 +所有插件行为都由带有 `SGLANG_FL_*` 前缀的环境变量控制。 + +有效配置优先级如下: + +```{code-block} text +environment variables > explicit YAML (`SGLANG_FL_CONFIG`) > platform auto-detected YAML > code defaults +``` + +环境变量可以覆盖显式 YAML 和平台 YAML 中的值。平台特定后端名称和运行时默认值取决于目标平台;当部署需要本通用指南未定义的可接受值列表时,请使用 `[TODO: needs confirmation]`。 + ## 第二层 — 融合算子调度 | 变量 | 默认值 | 描述 | |----------|---------|-------------| -| `SGLANG_FL_OOT_ENABLED` | `1` | 总开关:`0` 禁用第二层(保留第一层 ATen 激活) | +| `SGLANG_FL_OOT_ENABLED` | `1` | 总开关:`0` 禁用第二层(保留第一层 ATen) | | `SGLANG_FL_PREFER` | `flagos` | 全局后端偏好:`flagos`、`vendor`、`reference` | | `SGLANG_FL_PER_OP` | — | 逐算子后端优先级,例如 `rms_norm=vendor\|flagos;silu_and_mul=reference` | | `SGLANG_FL_OOT_BLACKLIST` | — | 跳过列出的算子,不进行 OOT 调度(逗号分隔的类名) | | `SGLANG_FL_OOT_WHITELIST` | — | 仅调度列出的算子(与 BLACKLIST 互斥) | -| `SGLANG_FL_STRICT` | `0` | `1` = 禁用回退(首选后端不可用时报错) | -| `SGLANG_FL_DENY_VENDORS` | — | 拒绝特定厂商(逗号分隔,例如 `cuda,ascend`) | +| `SGLANG_FL_STRICT` | `0` | `1` 禁用回退;首选或有序后端不可用时会报错 | +| `SGLANG_FL_DENY_VENDORS` | — | 拒绝指定厂商(逗号分隔) | | `SGLANG_FL_ALLOW_VENDORS` | — | 仅允许列出的厂商(逗号分隔) | | `SGLANG_FL_DISPATCH_LOG` | — | 调度日志文件路径(记录哪些算子被拦截) | +| `SGLANG_FL_DISPATCH_DEBUG` | [TODO: needs confirmation] | 启用额外调度诊断;可接受值为 [TODO: needs confirmation] | ## 第一层 — ATen 替换(FlagGems) @@ -34,15 +44,17 @@ | 变量 | 默认值 | 描述 | |----------|---------|-------------| -| `SGLANG_FL_DIST_BACKEND` | `nccl` | 后端:`nccl` / `hccl` / `flagcx` | -| `FLAGCX_PATH` | — | FlagCX 安装路径(设置后默认使用 `flagcx` 后端) | +| `SGLANG_FL_DIST_BACKEND` | 通用参考表中为 `nccl` | 分布式后端选择;平台运行时默认值可能映射到其他后端,且未设置显式覆盖时,`FLAGCX_PATH` 可以选择 FlagCX | +| `FLAGCX_PATH` | — | FlagCX 安装路径;使用 FlagCX collective communication 时必需 | + +支持的运行时选项取决于平台和已安装库。对于目标运行时未覆盖的平台特定可接受值,请使用 `[TODO: needs confirmation]`。 ## 系统 / 调试 | 变量 | 默认值 | 描述 | |----------|---------|-------------| | `SGLANG_FL_CONFIG` | — | YAML 配置文件路径(覆盖平台自动检测) | -| `SGLANG_FL_PLATFORM` | (自动) | 强制指定平台:`cuda`、`ascend`(覆盖自动检测) | +| `SGLANG_FL_PLATFORM` | (自动) | 强制指定平台;可接受值为 [TODO: needs confirmation] | | `SGLANG_FL_LOG_LEVEL` | `INFO` | 调度系统日志级别:`DEBUG`、`INFO`、`WARNING`、`ERROR` | | `SGLANG_PLUGINS` | (全部) | SGLang 内置:筛选要加载的插件(逗号分隔) | diff --git a/docs/sglang_plugin_fl_zh/dispatch_user_guide/dispatch-through-yaml-file.md b/docs/sglang_plugin_fl_zh/dispatch_user_guide/dispatch-through-yaml-file.md index 8372d1c557..13a7bb77b5 100644 --- a/docs/sglang_plugin_fl_zh/dispatch_user_guide/dispatch-through-yaml-file.md +++ b/docs/sglang_plugin_fl_zh/dispatch_user_guide/dispatch-through-yaml-file.md @@ -1,18 +1,23 @@ # 通过 YAML 配置文件进行调度 -插件附带一个示例配置文件 `config/sample.yaml`,包含所有可用选项。复制并自定义: + +插件在 `sglang_fl/dispatch/config/` 下提供平台 YAML 默认配置。可用的平台文件包括 `ascend.yaml`、`gcu.yaml`、`hygon.yaml`、`iluvatar.yaml`、`kunlunxin.yaml`、`musa.yaml`、`nvidia.yaml` 和 `tsingmicro.yaml`。这些文件提供默认调度策略;它们不是厂商验证矩阵。 -```{code-block} shell -# 复制示例配置 -cp $(python -c "from sglang_fl.config import _CONFIG_DIR; print(_CONFIG_DIR / 'sample.yaml')") my_config.yaml +当您需要覆盖平台默认值时,请创建显式 YAML 文件: -# 根据需要编辑,然后使用它启动 +```{code-block} shell SGLANG_FL_CONFIG=./my_config.yaml python -m sglang.launch_server \ --model-path Qwen/Qwen2.5-0.5B-Instruct \ --port 30000 --disable-piecewise-cuda-graph ``` -如果未设置 `SGLANG_FL_CONFIG`,插件使用合理的默认值(在 CUDA 上等同于 `prefer: flagos`)。仅当需要自定义行为时才需要 YAML 文件。 +配置优先级如下: + +```{code-block} text +environment variables > explicit YAML (`SGLANG_FL_CONFIG`) > platform auto-detected YAML > code defaults +``` + +如果未提供显式 YAML,则使用检测到的平台 YAML 和代码默认值。环境变量可以覆盖任一 YAML 来源。 ## 配置字段 @@ -20,34 +25,43 @@ SGLANG_FL_CONFIG=./my_config.yaml python -m sglang.launch_server \ # 全局后端偏好:flagos | vendor | reference prefer: flagos -# 逐算子后端优先级(有序列表,第一个可用的胜出) +# 逐算子后端优先级(有序列表,第一个可用者胜出) op_backends: rms_norm: [vendor, flagos, reference] silu_and_mul: [flagos, vendor, reference] -# 第二层要跳过的融合算子(回退到 SGLang 原生 CUDA) -# 可用:SiluAndMul、RMSNorm、RotaryEmbedding +# 要跳过的第二层融合算子(回退到 SGLang 原生路径) oot_blacklist: - RotaryEmbedding -# 第一层要从 FlagGems Triton 替换中排除的 ATen 算子 +# 要从 FlagGems Triton 替换中排除的第一层 ATen 算子 flagos_blacklist: - mul - sub + +# 可选厂商过滤器 +allow_vendors: [] +deny_vendors: [] + +# 为 true 时禁用回退 +strict: false ``` | 字段 | 描述 | |-------|-------------| | `prefer` | 全局后端偏好:`flagos`、`vendor`、`reference` | -| `op_backends` | 逐算子有序后端列表(第一个可用的胜出,可列出 1–3 个后端) | -| `oot_blacklist` | 第二层要从 OOT 调度中跳过的融合算子(回退到 SGLang 原生 CUDA) | -| `flagos_blacklist` | 第一层要从 FlagGems 替换中排除的 ATen 算子(回退到 PyTorch 原生) | +| `op_backends` | 逐算子有序后端列表;选择第一个可用且被允许的后端 | +| `oot_blacklist` | 要从 OOT 调度中跳过的第二层融合算子 | +| `flagos_blacklist` | 要从 FlagGems 替换中排除的第一层 ATen 算子 | +| `allow_vendors` | 可选厂商允许列表;只有列出的厂商符合条件 | +| `deny_vendors` | 可选厂商拒绝列表 | +| `strict` | 启用后禁用回退;确切 YAML 布尔语法为 [TODO: needs confirmation] | ## 常用配置方案 -每个方案展示一个 YAML 配置和预期的调度结果。使用[调度日志](/dispatch_user_guide/debugg-and-diagonostics.md)进行验证。 +每个方案展示一个 YAML 配置和预期的调度结果。使用[调度日志](debugg-and-diagonostics.md)进行验证。 -### 1. 跳过 RotaryEmbedding 的 OOT 调度(回退到 SGLang 原生 CUDA) +### 1. 跳过 RotaryEmbedding 的 OOT 调度(回退到 SGLang 原生路径) ```yaml # my_config.yaml @@ -67,7 +81,7 @@ op_backends: rms_norm: [vendor, flagos, reference] ``` -预期调度日志:`RMSNorm → vendor(vendor.nvidia)`,`SiluAndMul → flagos(flagos)`。 +预期调度日志:根据当前平台和厂商过滤器,选择第一个可用且被允许的后端。 ### 3. 所有算子使用纯 PyTorch reference(适用于精度调试) @@ -76,4 +90,4 @@ op_backends: prefer: reference ``` -预期调度日志:所有算子 → `reference(reference)`。 +预期调度日志:在可用时选择 reference 实现。 diff --git a/docs/sglang_plugin_fl_zh/dispatch_user_guide/dispatch-user-guide.md b/docs/sglang_plugin_fl_zh/dispatch_user_guide/dispatch-user-guide.md index a9748deebc..ecea4514af 100644 --- a/docs/sglang_plugin_fl_zh/dispatch_user_guide/dispatch-user-guide.md +++ b/docs/sglang_plugin_fl_zh/dispatch_user_guide/dispatch-user-guide.md @@ -1,8 +1,9 @@ # 算子调度用户指南 -调度系统提供三层算子替换。您可以独立且灵活地控制每一层。 + +调度系统通过 YAML 文件和环境变量提供算子替换与分布式通信配置。您可以独立控制替换层,并配置可感知平台的通信后端。 -调度系统支持 YAML 配置和环境变量进行细粒度控制。环境变量优先级高于 YAML 配置。 +调度系统同时支持 YAML 配置和环境变量,以便进行细粒度控制。环境变量优先于 YAML 配置。 优先级链如下: @@ -10,8 +11,6 @@ SGLANG_FL_* 环境变量 > YAML 配置(SGLANG_FL_CONFIG)> 平台自动检测 YAML > 代码默认值 ``` - - ```{toctree} :maxdepth: 2 diff --git a/docs/sglang_plugin_fl_zh/dispatch_user_guide/vendor-integration.md b/docs/sglang_plugin_fl_zh/dispatch_user_guide/vendor-integration.md index 1c1f74174d..bd45b151da 100644 --- a/docs/sglang_plugin_fl_zh/dispatch_user_guide/vendor-integration.md +++ b/docs/sglang_plugin_fl_zh/dispatch_user_guide/vendor-integration.md @@ -7,15 +7,10 @@ cp -r sglang_fl/dispatch/backends/vendor/template/ \ sglang_fl/dispatch/backends/vendor/my_chip/ ``` -下表列出了已有的厂商: + +厂商特定的框架选择、运行时镜像、验证状态、安装命令和适配流程维护在集中式页面:[集中式厂商 / 框架 / 镜像选择页面](https://flagos.io/resourcedownload?lang=en)。下面的通用契约并不意味着每个后端都支持所有算子或模型。 -| 厂商 | 目录 | 硬件检测 | -| :--- | :--- | :--- | -| NVIDIA CUDA | `vendor/cuda/` | `sgl_kernel` 可导入 | -| 华为昇腾 | `vendor/ascend/` | `torch_npu` 可导入 | -| 模板 | `vendor/template/` | 始终为 False(仅作参考) | - -要集成新的厂商,需要实现两个文件: +要集成新厂商,请实现下面所述的后端契约和注册逻辑。示例中的三个算子只是最小示例,并非 v0.2.0 的完整算子范围。 ## 1. 后端类(my_chip.py) @@ -115,9 +110,18 @@ def register_builtins(registry) -> None: | `rms_norm` | `fn(obj, x: Tensor, residual: Optional[Tensor] = None) -> Tensor \| tuple[Tensor, Tensor]` | | `rotary_embedding` | `fn(obj, query, key, cos, sin, position_ids, rotary_interleaved=False, inplace=True) -> tuple[Tensor, Tensor]` | - `obj` 参数提供对层属性的访问(`obj.weight`、`obj.variance_epsilon` 等)。这些属性名称在 SGLang 和 vLLM 之间完全相同,因此同一实现可同时用于两个框架。 ## 厂商后端自动发现 -插件在启动时扫描 `dispatch/backends/vendor/*/register_ops.py`。如果 is_available() 返回 True,该厂商的算子即被注册。无需修改其他文件。 +插件在启动时扫描 `dispatch/backends/vendor/*/register_ops.py`。如果 `is_available()` 返回 True,该厂商的算子即被注册。无需修改其他文件。 + + +如有需要,可选集成钩子可以添加平台特定行为: + +- `patch.py` 可以应用公开后端或传输路径所需的框架补丁。 +- `register_platform.py` 可以注册平台标识和运行时行为。 +- `sglang_fl/dispatch/config/` 下的平台 YAML 可以为检测到的平台提供默认调度策略。 + +这些文件扩展了通用集成契约;它们不构成普遍的厂商支持或验证声明。详细适配流程仍维护在集中式页面:[集中式厂商 / 框架 / 镜像选择页面](https://flagos.io/resourcedownload?lang=en)。 + diff --git a/docs/sglang_plugin_fl_zh/getting_started/getting-started.md b/docs/sglang_plugin_fl_zh/getting_started/getting-started.md index cf1a990e2c..50b4b937a8 100644 --- a/docs/sglang_plugin_fl_zh/getting_started/getting-started.md +++ b/docs/sglang_plugin_fl_zh/getting_started/getting-started.md @@ -7,6 +7,7 @@ requirements.md install.md -run-inference-task.md + +quick-run-inference-task.md ``` diff --git a/docs/sglang_plugin_fl_zh/getting_started/install.md b/docs/sglang_plugin_fl_zh/getting_started/install.md index 209a006ae7..069f7cd74c 100644 --- a/docs/sglang_plugin_fl_zh/getting_started/install.md +++ b/docs/sglang_plugin_fl_zh/getting_started/install.md @@ -1,28 +1,25 @@ # 安装 sglang-plugin-FL -## Docker 镜像(推荐) - -v0.1.0-rc2 预构建 Docker 镜像: - -| 平台 | 镜像 | 内容 | -|----------|-------|----------| -| NVIDIA GPU(双节点) | `harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-nvidia-dual` | sglang 0.5.11, flag_gems 5.3.0rc2, torch 2.11.0, triton 3.6.0 | -| NVIDIA GPU(单节点) | `harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-nvidia-single` | sglang 0.5.11, flag_gems 5.3.0rc2, torch 2.11.0, triton 3.6.0 | -| 摩尔线程 MUSA | `harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-musa` | sglang 0.5.12, torch 2.9.0, flag_gems 5.0.2 | -| 摩尔线程(SVT) | `harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-mthreads-svt` | sglang 0.5.11, flag_gems 5.3.0rc2, torch 2.9.0, triton 3.1.0 | -| 华为昇腾 | `harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-ascend` | sglang 0.5.12, flag_gems 5.0.2, CANN 8.5.0 | - -```bash -# NVIDIA 双节点(跨节点推理) -docker pull harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-nvidia-dual -# NVIDIA 单节点 -docker pull harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-nvidia-single -# 摩尔线程 MUSA -docker pull harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-musa -# 摩尔线程 SVT(全栈测试) -docker pull harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-mthreads-svt -# 华为昇腾 -docker pull harbor.baai.ac.cn/flagos21-release/sglang-plugin-fl:v0.1.0-rc2-ascend -``` - -双节点镜像支持跨节点大模型推理,单节点镜像支持单机推理。SVT 为全栈测试镜像。 + +## 运行时镜像与平台软件包 + +特定厂商的框架选择、运行时镜像、验证状态、安装命令和适配流程统一维护在集中页面:[厂商/框架/镜像选择集中页面](https://flagos.io/resourcedownload?lang=en)。安装或启动 sglang-plugin-FL 前,请使用该页面选择厂商、框架和镜像。 + +本页面有意不再维护完整镜像矩阵或平台专属软件包配方。 + + +## Empty mode + +Empty mode 是一种安装/运行时组装机制,适用于以 CUDA 为导向的 SGLang 依赖栈并非目标部署环境的平台。它避免将 CUDA 软件包视为通用依赖集,但它**不是**无设备模式。 + +目标平台仍需提供: + +- 厂商 torch; +- 驱动和固件; +- 设备运行时; +- 通信库; +- 平台注意力后端; +- sglang-plugin-FL 未覆盖的算子。 + +请使用集中厂商/框架/镜像选择页面获取平台专属镜像和依赖集:[厂商/框架/镜像选择集中页面](https://flagos.io/resourcedownload?lang=en)。 + diff --git a/docs/sglang_plugin_fl_zh/getting_started/quick-run-inference-task.md b/docs/sglang_plugin_fl_zh/getting_started/quick-run-inference-task.md index 22b44867ba..51566b4072 100644 --- a/docs/sglang_plugin_fl_zh/getting_started/quick-run-inference-task.md +++ b/docs/sglang_plugin_fl_zh/getting_started/quick-run-inference-task.md @@ -32,13 +32,17 @@ cd FlagCX && make USE_NVIDIA=1 export FLAGCX_PATH="$PWD" ``` + +厂商专属 Empty mode 安装、运行时镜像、框架软件包和验证前置条件请参阅厂商/框架/镜像选择集中页面:[厂商/框架/镜像选择集中页面](https://flagos.io/resourcedownload?lang=en)。Empty mode 不是无设备模式;目标平台必须提供其厂商 torch、驱动、固件、设备运行时、通信库、注意力后端,以及未覆盖的算子。 + + ## 下载模型 ```{code-block} shell # 用于快速测试的小模型(单 GPU) huggingface-cli download Qwen/Qwen2.5-0.5B-Instruct -# 用于多 GPU 的大模型(tp=8) +# 用于多 GPU 的较大模型(tp=8) huggingface-cli download Qwen/Qwen2.5-14B-Instruct ``` @@ -52,7 +56,7 @@ HF_ENDPOINT=https://hf-mirror.com huggingface-cli download Qwen/Qwen2.5-0.5B-Ins ## 运行推理任务 -### 1. 启动 SGLang 服务器 +### 1. 启动 sglang 服务器 #### 单 GPU @@ -63,7 +67,7 @@ python -m sglang.launch_server \ --disable-piecewise-cuda-graph ``` -#### 多 GPU 张量并行 +#### 使用张量并行的多 GPU ```{code-block} shell python -m sglang.launch_server \ @@ -73,9 +77,25 @@ python -m sglang.launch_server \ ``` ```{note} -FlagGems Triton 内核包含 `logging.Logger` 调用,与 `torch.compile`(SGLang 的分段 CUDA 图使用)不兼容。启动服务器时请始终使用 `--disable-piecewise-cuda-graph`。常规 CUDA 图捕获可正常工作。 +FlagGems Triton 内核包含 `logging.Logger` 调用,与 `torch.compile`(SGLang 的分段 CUDA 图使用该功能)不兼容。启动服务器时请始终使用 `--disable-piecewise-cuda-graph`。常规 CUDA 图捕获可正常工作。 ``` + +### 多节点与流水线并行 + +对于多节点推理,请为目标平台配置分布式后端和网络接口,然后使用相应的 SGLang 张量并行和流水线并行参数。仓库中的多节点示例是起点,而不是通用硬件配方;请将地址、设备可见性变量、通信路径和接口名称替换为目标部署环境中的值。 + +流水线并行工作流可通过 `CommunicatorFL` 使用 FlagCX 或 `torch.distributed`。后端选择请参阅调度环境变量指南。平台专属镜像、框架构建版本和验证状态统一维护在集中页面:[厂商/框架/镜像选择集中页面](https://flagos.io/resourcedownload?lang=en)。 + +### Qwen3.6 MTP + +v0.2.0 示例包含 Qwen3.6 多词元预测(MTP)工作流覆盖。请使用模型支持的 SGLang 启动参数以及目标平台已验证的运行时选择;本页面不宣称提供通用模型或硬件支持。 + +### 吞吐量基准测试 + +推理路径工作正常后,可使用仓库中的 serving benchmark 工具。请在不同输入/输出长度和请求速率下比较吞吐量、首词元延迟和解码延迟。不要从通用工作流中推断普适性能特征。 + + ### 2. 发送请求 服务器就绪后(显示 `The server is fired up and ready to roll`),发送请求: diff --git a/docs/sglang_plugin_fl_zh/getting_started/requirements.md b/docs/sglang_plugin_fl_zh/getting_started/requirements.md index b2104fc6b3..314b8d51a1 100644 --- a/docs/sglang_plugin_fl_zh/getting_started/requirements.md +++ b/docs/sglang_plugin_fl_zh/getting_started/requirements.md @@ -2,29 +2,32 @@ ## 软件要求 -sglang-plugin-FL 需要以下软件版本。 + +所需运行时栈取决于所选厂商、框架和镜像。平台专属 SDK、框架构建版本、镜像、验证状态和软件包版本请参阅厂商/框架/镜像选择集中页面:[厂商/框架/镜像选择集中页面](https://flagos.io/resourcedownload?lang=en)。 -| 软件包 | 版本 | +sglang-plugin-FL 的通用要求如下: + +| 组件 | 要求 | |---------|---------| -| SGLang | 0.5.11 | -| sglang-kernel | 0.4.2 | -| PyTorch | 2.11.0+cu130 | -| Triton | 3.6.0 | -| FlagGems | 4.2.1rc0 | -| flashinfer | 0.6.8.post1 | -| Python | 3.12 | -| CUDA | 13.0 | +| SGLang | 支持通过插件入口点加载的兼容 SGLang 运行时 | +| sglang-plugin-FL | 已安装在目标运行时环境中 | +| FlagGems | 启用时,用于第 1 层 ATen 替换以及基于 FlagGems 的融合实现 | +| FlagCX | 使用 FlagCX 集合通信时需要 | +| 厂商运行时栈 | 厂商 torch、驱动、固件、设备运行时、通信库、注意力后端,以及未被插件覆盖的算子 | + + +## Empty mode 边界 + +Empty mode 不是无设备模式。它改变安装/运行时组装方式,使厂商专属环境能够提供自己的运行时栈,而不是继承以 CUDA 为导向的依赖集。目标平台仍需提供厂商 torch、驱动、固件、设备运行时、通信库、平台注意力后端,以及插件未覆盖的算子。 + ## 硬件要求 -- 支持 CUDA 13.0 的 NVIDIA GPU,或 +- 支持 CUDA 的 NVIDIA GPU,或 - 配备 CANN 工具包的华为昇腾 NPU,或 -- 其他支持相应厂商 SDK 的硬件 +- 其他配备相应厂商 SDK 的受支持硬件 ## 已验证的模型 -| 模型 | TP | 状态 | -|-------|-----|--------| -| Qwen3.6-27B (混合注意力 + FLA + MoE) | tp=1 | 已验证 | -| Qwen3.6-35B-A3B (MoE, 256 专家) | tp=1 | 已验证 | -| Qwen2.5-14B-Instruct | tp=8 | 已验证 | + +详细的模型、量化、平台和验证状态请参阅集中页面:[厂商/框架/镜像选择集中页面](https://flagos.io/resourcedownload?lang=en)。请勿将本页面视为完整的厂商支持矩阵。 diff --git a/docs/sglang_plugin_fl_zh/index.md b/docs/sglang_plugin_fl_zh/index.md index cf48d1d6f0..d91aad0be5 100644 --- a/docs/sglang_plugin_fl_zh/index.md +++ b/docs/sglang_plugin_fl_zh/index.md @@ -42,13 +42,13 @@ ::: :::{grid-item-card} {octicon}`code;1.5em;sd-mr-1` API 参考 -:link: reference/dispatch-api-reference +:link: reference/dispatch-reference-and-example :link-type: doc 算子调度系统和配置选项的 API 参考。 +++ -[了解更多 »](reference/dispatch-api-reference.md) +[了解更多 »](reference/dispatch-reference-and-example.md) ::: :::: @@ -78,5 +78,5 @@ dispatch_user_guide/dispatch-user-guide.md :maxdepth: 5 :hidden: -reference/dispatch-api-reference.md +reference/dispatch-reference-and-example.md ``` diff --git a/docs/sglang_plugin_fl_zh/overview/features.md b/docs/sglang_plugin_fl_zh/overview/features.md index 15754c6c4f..b9e6f12615 100644 --- a/docs/sglang_plugin_fl_zh/overview/features.md +++ b/docs/sglang_plugin_fl_zh/overview/features.md @@ -2,7 +2,8 @@ SGLang 的推理引擎依赖 NVIDIA 专用组件:flashinfer 用于注意力计算,sgl_kernel 用于融合 CUDA 内核,NCCL 用于分布式通信。在其他硬件(华为昇腾、寒武纪 MLU、Iluvatar 等)上运行原本需要对源码进行侵入式修改。 -本插件通过三个层次的替换提供了非侵入式的适配层: + +本插件通过三个层次的替换与扩展提供非侵入式适配层: ## 第一层 — ATen 算子 @@ -10,15 +11,18 @@ SGLang 的推理引擎依赖 NVIDIA 专用组件:flashinfer 用于注意力计 ## 第二层 — SGLang 融合内核 -通过 HookRegistry AROUND 钩子拦截 SGLang 的自定义融合算子(SiluAndMul、RMSNorm、RotaryEmbedding),经过标准化调度系统(与 vllm-plugin-FL 对齐)路由,选择最佳可用后端: +通过 HookRegistry AROUND 钩子拦截 SGLang 的自定义融合算子,经过标准化调度系统(与 vllm-plugin-FL 对齐)进行路由,以选择最佳可用后端: -- **FlagGems** — 基于 Triton 的实现(默认,最高优先级) -- **Vendor** — 芯片原生实现(例如 CUDA sgl_kernel、昇腾 CANN) +- **FlagGems** — 基于 Triton 的实现 +- **Vendor** — 芯片原生实现 - **Reference** — 纯 PyTorch 回退实现 +调度策略会综合全局后端偏好、逐算子后端顺序、后端可用性、厂商允许/拒绝过滤、严格模式和回退行为。请勿将上述列表理解为所有算子的固定优先级顺序。 + ## 第三层 — 分布式通信 -用 CommunicatorFL(基于 FlagCX 或 torch.distributed)替换基于 NCCL 的集合通信,支持任意硬件上的多卡推理。支持 all_reduce、all_gather、reduce_scatter、send 和 recv 操作。 +用 CommunicatorFL(基于 FlagCX 或 torch.distributed)替换基于 NCCL 的集合通信,从而在不同硬件后端上支持多卡推理。支持 all_reduce、all_gather、reduce_scatter、send 和 recv 操作,并支持平台感知的后端选择和流水线并行通信。 + ![alt text](../assets/sglang-plugin-fl-arch.png) diff --git a/docs/sglang_plugin_fl_zh/overview/how-plugin-works.md b/docs/sglang_plugin_fl_zh/overview/how-plugin-works.md index c8b686f323..733f7087d4 100644 --- a/docs/sglang_plugin_fl_zh/overview/how-plugin-works.md +++ b/docs/sglang_plugin_fl_zh/overview/how-plugin-works.md @@ -18,6 +18,7 @@ sglang_fl = "sglang_fl:activate_platform" 核心机制使用 `MultiPlatformOp.dispatch_forward()` 上的 AROUND 钩子,结合标准化调度系统: + ```{code-block} python dispatch_forward() 被调用(例如 RMSNorm) → AROUND 钩子拦截 @@ -28,12 +29,17 @@ dispatch_forward() 被调用(例如 RMSNorm) rms_norm_bridge(self, x, residual, post_residual_addition) → 桥接函数处理 SGLang 特定参数(post_residual_addition → 合并到 residual) → 桥接函数调用 dispatch.call_op("rms_norm", obj, x, residual) - → OpManager 通过策略解析最佳实现(flagos > vendor > reference) - → 调用选中的后端:rms_norm_flaggems(obj, x, residual) + → OpManager 请求 SelectionPolicy 提供候选后端顺序 + → 策略综合全局偏好、逐算子顺序、可用性、厂商过滤和严格模式 + → OpManager 选择第一个可用实现并缓存结果,或在允许时执行回退 + → 调用选中的后端实现 ``` - 桥接层将框架特定参数与标准化算子签名解耦。厂商后端只需实现标准签名——同一实现可同时用于 sglang-plugin-FL 和 vllm-plugin-FL。 + +严格模式会禁用回退:如果首选后端或显式指定顺序中的后端无法使用,调度会报告错误,而不是静默选择其他后端。在非严格模式下,不可用后端会被过滤,随后运行下一个允许的候选后端。 + + ## 调度架构(与 vllm-plugin-FL 共享) ```{code-block} python @@ -51,14 +57,11 @@ dispatch_forward() 被调用(例如 RMSNorm) ┌────────────────┼────────────────┐ ▼ ▼ ▼ ┌─────────────┐ ┌───────────┐ ┌──────────────┐ - │ DEFAULT │ │ VENDOR │ │ REFERENCE │ - │ (FlagGems) │ │ (Ascend/ │ │ (PyTorch) │ - │ priority=150│ │ CUDA) │ │ priority=50 │ - │ │ │ priority= │ │ │ - │ │ │ 100 │ │ │ + │ FLAGOS │ │ VENDOR │ │ REFERENCE │ + │ (FlagGems) │ │ (chip- │ │ (PyTorch) │ + │ │ │ native) │ │ │ └─────────────┘ └───────────┘ └──────────────┘ ``` - 芯片厂商为两个框架实现**相同的后端接口**。唯一的框架特定代码是桥接层,由插件维护。 ## ATen 替换 @@ -70,3 +73,11 @@ dispatch_forward() 被调用(例如 RMSNorm) → _AtenOnlyFilter 确保只记录 flag_gems.ops.* 调用 (排除第二层 flagos 实现中触发的内部 FlagGems 调用) ``` + + +## 空模式边界 + +空模式是一种安装与运行时组装机制,适用于以 CUDA 为导向的依赖集并非部署环境的目标平台。它不是无设备模式。目标平台仍需提供厂商 torch、驱动、固件、设备运行时、通信库、平台注意力后端以及插件未覆盖的算子。 + +有关特定平台的运行时和镜像选择,请使用集中式厂商/框架/镜像选择页面:[集中式厂商/框架/镜像选择页面](https://flagos.io/resourcedownload?lang=en)。 + diff --git a/docs/sglang_plugin_fl_zh/overview/project-structure.md b/docs/sglang_plugin_fl_zh/overview/project-structure.md index d3388c837e..0d2cbd8852 100644 --- a/docs/sglang_plugin_fl_zh/overview/project-structure.md +++ b/docs/sglang_plugin_fl_zh/overview/project-structure.md @@ -1,5 +1,6 @@ # 项目结构 + ```{code-block} python sglang_fl/ ├── pyproject.toml # 包配置 + entry_points 注册 @@ -11,11 +12,6 @@ sglang_fl/ │ ├── communicator.py # CommunicatorFL(FlagCX / torch.distributed 封装) │ └── device_communicators/ │ └── flagcx.py # FlagCX 专用通信器 - ├── config/ - │ ├── __init__.py # YAML 配置加载器,支持平台自动检测 - │ ├── sample.yaml # 完整示例配置,包含所有选项的文档说明 - │ ├── nvidia.yaml # NVIDIA CUDA 平台默认配置 - │ └── ascend.yaml # 昇腾平台默认配置(含黑名单) └── dispatch/ # 算子调度系统(与 vllm-plugin-FL 对齐) ├── __init__.py # 公共 API:call_op()、resolve_op() ├── types.py # OpImpl、BackendImplKind、BackendPriority @@ -25,6 +21,15 @@ sglang_fl/ ├── builtin_ops.py # 注册编排器 ├── ops.py # FLBackendBase ABC(算子签名定义) ├── logger_manager.py # 使用 SGLANG_FL_LOG_LEVEL 进行日志记录 + ├── config/ # 平台 YAML 默认配置和调度配置 + │ ├── ascend.yaml + │ ├── gcu.yaml + │ ├── hygon.yaml + │ ├── iluvatar.yaml + │ ├── kunlunxin.yaml + │ ├── musa.yaml + │ ├── nvidia.yaml + │ └── tsingmicro.yaml ├── bridge/ # SGLang ↔ 调度参数转换 │ ├── __init__.py │ ├── silu_and_mul.py # forward_cuda(self, x) → call_op("silu_and_mul", obj, x) @@ -41,8 +46,10 @@ sglang_fl/ │ ├── register_ops.py │ └── impl/ # activation.py、normalization.py、rotary.py └── vendor/ # VENDOR 后端(自动发现) - ├── ascend/ # 华为昇腾 NPU(torch_npu) - ├── cuda/ # NVIDIA CUDA(sgl_kernel) + ├── / # 厂商特定后端包 └── template/ # 新厂商模板 ``` + +平台 YAML 文件为检测到的平台提供默认调度策略。它们不是厂商验证矩阵。厂商特定的框架选择、运行时镜像、验证状态、安装命令和适配流程统一在集中式页面维护:[集中式厂商/框架/镜像选择页面](https://flagos.io/resourcedownload?lang=en)。 + diff --git a/docs/sglang_plugin_fl_zh/reference/dispatch-reference-and-example.md b/docs/sglang_plugin_fl_zh/reference/dispatch-reference-and-example.md index 0ef34c1ac3..60aafeafa5 100644 --- a/docs/sglang_plugin_fl_zh/reference/dispatch-reference-and-example.md +++ b/docs/sglang_plugin_fl_zh/reference/dispatch-reference-and-example.md @@ -1,6 +1,16 @@ # 环境变量参考与示例 -本页记录了 sglang-plugin-FL 的环境变量参考和示例。 +本页记录 sglang-plugin-FL 的环境变量参考和示例。 + + +有效配置优先级如下: + +```{code-block} text +environment variables > explicit YAML (`SGLANG_FL_CONFIG`) > platform auto-detected YAML > code defaults +``` + +关于厂商 / 框架 / 镜像选择、平台特定运行时软件包、镜像和验证状态,请参阅[集中式厂商 / 框架 / 镜像选择页面](https://flagos.io/resourcedownload?lang=en)。 + ## 环境变量 — 完整参考 @@ -18,7 +28,6 @@ | `SGLANG_FL_ALLOW_VENDORS` | — | 仅允许列出的厂商(逗号分隔) | | `SGLANG_FL_DISPATCH_LOG` | — | 调度日志文件路径(记录哪些算子被拦截) | - ### 第一层 — ATen 替换(FlagGems) | 变量 | 默认值 | 描述 | @@ -36,15 +45,17 @@ | 变量 | 默认值 | 描述 | |----------|---------|-------------| -| `SGLANG_FL_DIST_BACKEND` | `nccl` | 后端:`nccl` / `hccl` / `flagcx` | -| `FLAGCX_PATH` | — | FlagCX 安装路径(设置后默认使用 `flagcx` 后端) | +| `SGLANG_FL_DIST_BACKEND` | 通用参考表中为 `nccl` | 分布式后端选择;平台运行时默认值可能映射到其他后端,且未设置显式覆盖时,`FLAGCX_PATH` 可以选择 FlagCX | +| `FLAGCX_PATH` | — | FlagCX 安装路径;使用 FlagCX collective communication 时必需 | + +支持的运行时选项取决于平台和已安装库。对于目标运行时未覆盖的平台特定可接受值,请使用 `[TODO: needs confirmation]`。 ### 系统 / 调试 | 变量 | 默认值 | 描述 | |----------|---------|-------------| | `SGLANG_FL_CONFIG` | — | YAML 配置文件路径(覆盖平台自动检测) | -| `SGLANG_FL_PLATFORM` | (自动) | 强制指定平台:`cuda`、`ascend`(覆盖自动检测) | +| `SGLANG_FL_PLATFORM` | (自动) | 强制指定平台;可接受值为 [TODO: needs confirmation] | | `SGLANG_FL_LOG_LEVEL` | `INFO` | 调度系统日志级别:`DEBUG`、`INFO`、`WARNING`、`ERROR` | | `SGLANG_PLUGINS` | (全部) | SGLang 内置:筛选要加载的插件(逗号分隔)。无需使用——插件在 `pip install` 后自动发现 | diff --git a/docs/sglang_plugin_fl_zh/release_notes/release-notes.md b/docs/sglang_plugin_fl_zh/release_notes/release-notes.md index 484d961b28..298fbc7401 100644 --- a/docs/sglang_plugin_fl_zh/release_notes/release-notes.md +++ b/docs/sglang_plugin_fl_zh/release_notes/release-notes.md @@ -1,5 +1,22 @@ # 发布说明 + +## v0.2.0 + +本次发布扩展了 sglang-plugin-FL 的调度、平台配置和运行时配置能力,完成了从 `v0.1.0` 到 `v0.2.0` 的演进。 + +- 新增功能 + - 通过 `sglang_fl/dispatch/config/` 下的平台 YAML 默认配置扩展平台配置。 + - 扩展调度策略控制,包括全局后端偏好、逐算子后端顺序、厂商允许/拒绝过滤、严格模式、回退行为以及调度缓存失效。 + - 为厂商特定运行时栈提供空模式安装与运行时组装边界。空模式不是无设备模式;目标平台仍需提供厂商 torch、驱动、固件、设备运行时、通信库以及插件未覆盖的算子。 + - 提供多节点和流水线并行推理示例、Qwen3.6 MTP 工作流覆盖以及吞吐量基准测试基础设施。 + +- 文档范围 + + - 厂商特定的框架选择、运行时镜像、验证状态、安装命令和适配流程统一在本英文插件文档之外进行维护:[集中式厂商/框架/镜像选择页面](https://flagos.io/resourcedownload?lang=en)。 + - 本文档仅保留通用插件架构、调度配置和工作流形态,不维护完整的厂商支持或镜像矩阵。 + + ## v0.1.0 diff --git a/docs/pytorch_plugin_fl_en/architecture/distributed.md b/docs/torch_fl_en/architecture/distributed.md similarity index 91% rename from docs/pytorch_plugin_fl_en/architecture/distributed.md rename to docs/torch_fl_en/architecture/distributed.md index 5c3c6877d0..a0ad1bfa1e 100644 --- a/docs/pytorch_plugin_fl_en/architecture/distributed.md +++ b/docs/torch_fl_en/architecture/distributed.md @@ -1,6 +1,6 @@ # Distributed Collectives -PyTorch-Plugin-FL provides distributed support for the `flagos` device through `ProcessGroupFlagOS`, a native `torch.distributed.ProcessGroup` subclass. It is registered at import, so `torch.distributed.init_process_group("flagos")` works with no `torch.distributed.*` monkeypatching. +Torch-FL provides distributed support for the `flagos` device through `ProcessGroupFlagOS`, a native `torch.distributed.ProcessGroup` subclass. It is registered at import, so `torch.distributed.init_process_group("flagos")` works with no `torch.distributed.*` monkeypatching. ## How it works @@ -45,7 +45,7 @@ flagos_dist.move_buffers_to_device(model, "flagos:0") ### DDP -At import, PyTorch-Plugin-FL patches `torch.nn.parallel.DistributedDataParallel.__init__`. When the model lives on a `flagos` device, the patch: +At import, Torch-FL patches `torch.nn.parallel.DistributedDataParallel.__init__`. When the model lives on a `flagos` device, the patch: - forces the Python reducer, bypassing the C++ reducer's CUDA assertion, and - replaces the default gradient-accumulation hook (which uses functional collectives that have no `privateuseone` dispatch) with a version that goes through `dist.all_reduce` and therefore through `ProcessGroupFlagOS`. diff --git a/docs/pytorch_plugin_fl_en/architecture/profiler.md b/docs/torch_fl_en/architecture/profiler.md similarity index 97% rename from docs/pytorch_plugin_fl_en/architecture/profiler.md rename to docs/torch_fl_en/architecture/profiler.md index bd0c25de0c..0b31361dbe 100644 --- a/docs/pytorch_plugin_fl_en/architecture/profiler.md +++ b/docs/torch_fl_en/architecture/profiler.md @@ -64,7 +64,7 @@ Two warnings are deliberately not gated by `FLAGOS_TRACE` — an empty linked-ac | 6 | `test_runtime_names_come_from_cbid` | Runtime event names are decoded from the callback id | | 7 | `test_capture_window_containment` | No device or runtime event escapes the capture window | -The baseline lives in `tests/data/profiler_cuda_baseline.json`. Assertion 2 is deliberately stricter than upstream `torch.cuda`: it is a PyTorch-Plugin-FL invariant, kept because a regression into dangling flow halves is exactly the bug it guards against. +The baseline lives in `tests/data/profiler_cuda_baseline.json`. Assertion 2 is deliberately stricter than upstream `torch.cuda`: it is a Torch-FL invariant, kept because a regression into dangling flow halves is exactly the bug it guards against. ## Known gaps diff --git a/docs/pytorch_plugin_fl_en/architecture/torch-compile.md b/docs/torch_fl_en/architecture/torch-compile.md similarity index 98% rename from docs/pytorch_plugin_fl_en/architecture/torch-compile.md rename to docs/torch_fl_en/architecture/torch-compile.md index 686af2760c..7b4bf7c6ef 100644 --- a/docs/pytorch_plugin_fl_en/architecture/torch-compile.md +++ b/docs/torch_fl_en/architecture/torch-compile.md @@ -36,7 +36,7 @@ model = torch.compile(model, backend="flagos", options={"max_autotune": True}) - Its wheel is named `flagtree`, but the module it installs is `triton`. - Installing it uninstalls the official `triton` and takes its place. -- Inductor's own `import triton` therefore already resolves to FlagTree once it is installed, and nothing in PyTorch-Plugin-FL patches `sys.modules`. +- Inductor's own `import triton` therefore already resolves to FlagTree once it is installed, and nothing in Torch-FL patches `sys.modules`. The backend compiler is selected at FlagTree build time through `FLAGTREE_BACKEND` (unset for NVIDIA and AMD), not at runtime: the same Triton kernel code compiles for a different vendor backend. `is_flagtree_active()` detects a FlagTree build, and `FLAGOS_USE_FLAGTREE=1` asserts that the active Triton is FlagTree — it errors rather than silently compiling with stock Triton. diff --git a/docs/pytorch_plugin_fl_en/assets/images/pytorch-plugin-fl.png b/docs/torch_fl_en/assets/images/torch-fl.png similarity index 100% rename from docs/pytorch_plugin_fl_en/assets/images/pytorch-plugin-fl.png rename to docs/torch_fl_en/assets/images/torch-fl.png diff --git a/docs/pytorch_plugin_fl_en/getting_started/installation.md b/docs/torch_fl_en/getting_started/installation.md similarity index 99% rename from docs/pytorch_plugin_fl_en/getting_started/installation.md rename to docs/torch_fl_en/getting_started/installation.md index 026d19285e..59172ad955 100644 --- a/docs/pytorch_plugin_fl_en/getting_started/installation.md +++ b/docs/torch_fl_en/getting_started/installation.md @@ -128,7 +128,7 @@ pip install torch==2.10.0 --index-url https://download.pytorch.org/whl/cpu FLAGOS_ACCELERATOR=gcu pip install --no-build-isolation -v -e . ``` -Requires the TopsRider SDK (`libtopsrt.so` runtime and `libtopsaten.so` operator library). The build runs `scripts/codegen/codegen_gcu.py`, validating each operator against the `topsaten` symbols actually present in the installed `libtopsaten.so`; operators missing from the SDK are skipped with a warning. The `torch-gcu` wheel cannot be used alongside PyTorch-Plugin-FL, because it claims `PrivateUse1` for itself. +Requires the TopsRider SDK (`libtopsrt.so` runtime and `libtopsaten.so` operator library). The build runs `scripts/codegen/codegen_gcu.py`, validating each operator against the `topsaten` symbols actually present in the installed `libtopsaten.so`; operators missing from the SDK are skipped with a warning. The `torch-gcu` wheel cannot be used alongside Torch-FL, because it claims `PrivateUse1` for itself. ### Moore Threads MUSA diff --git a/docs/pytorch_plugin_fl_en/index.md b/docs/torch_fl_en/index.md similarity index 80% rename from docs/pytorch_plugin_fl_en/index.md rename to docs/torch_fl_en/index.md index a9e7394010..02b3da20ed 100644 --- a/docs/pytorch_plugin_fl_en/index.md +++ b/docs/torch_fl_en/index.md @@ -1,8 +1,8 @@ -# PyTorch-Plugin-FL +# Torch-FL -PyTorch-Plugin-FL (`torch_fl`) is a PyTorch device plugin for the FlagOS software stack. It exposes a single `flagos` device that routes operators among reusable native kernels, portable compiler kernels, vendor-native implementations, and explicit CPU fallback, so the same PyTorch program runs across accelerators without workload changes. +Torch-FL (`torch_fl`) is a PyTorch device plugin for the FlagOS software stack. It exposes a single `flagos` device that routes operators among reusable native kernels, portable compiler kernels, vendor-native implementations, and explicit CPU fallback, so the same PyTorch program runs across accelerators without workload changes. -![PyTorch-Plugin-FL architecture](assets/images/pytorch-plugin-fl.png) +![Torch-FL architecture](assets/images/torch-fl.png) ::::{grid} 1 2 2 3 :gutter: 1 1 1 2 @@ -11,7 +11,7 @@ PyTorch-Plugin-FL (`torch_fl`) is a PyTorch device plugin for the FlagOS softwar :link: overview/overview :link-type: doc -What PyTorch-Plugin-FL is, its design principles, capabilities, and component architecture. +What Torch-FL is, its design principles, capabilities, and component architecture. +++ [Learn more »](overview/overview.md) @@ -69,8 +69,6 @@ Per-platform capability validation, environment variables, and operation-routing :::: -## Project links - - **Repository**: [flagos-ai/Torch-FL](https://github.com/flagos-ai/Torch-FL) - **FlagGems**: [flagos-ai/FlagGems](https://github.com/flagos-ai/FlagGems) - **FlagTree**: [flagos-ai/FlagTree](https://github.com/flagos-ai/FlagTree) diff --git a/docs/pytorch_plugin_fl_en/overview/architecture.md b/docs/torch_fl_en/overview/architecture.md similarity index 92% rename from docs/pytorch_plugin_fl_en/overview/architecture.md rename to docs/torch_fl_en/overview/architecture.md index fc1700a375..98f85e52d6 100644 --- a/docs/pytorch_plugin_fl_en/overview/architecture.md +++ b/docs/torch_fl_en/overview/architecture.md @@ -1,8 +1,8 @@ # Architecture -PyTorch-Plugin-FL registers one PyTorch device and routes every operator that reaches it. +Torch-FL registers one PyTorch device and routes every operator that reaches it. -![PyTorch-Plugin-FL architecture](../assets/images/pytorch-plugin-fl.png) +![Torch-FL architecture](../assets/images/torch-fl.png) ```text PyTorch API @@ -18,7 +18,7 @@ accelerator runtime ## Device registration -At import, PyTorch-Plugin-FL claims the `PrivateUse1` dispatch key and publishes it under the name `flagos`: +At import, Torch-FL claims the `PrivateUse1` dispatch key and publishes it under the name `flagos`: 1. `torch.utils.rename_privateuse1_backend("flagos")` names the key. 2. `torch._register_device_module("flagos", flagos)` installs the device module, so `torch.flagos.*` works. @@ -69,7 +69,7 @@ Two constraints are worth knowing when debugging an import: the vendor libtorch Operator implementations are reached through one dispatch key and one routing table: - The generated bindings under `csrc/aten/generated/` provide the CUDA-boxing kernels, the FlagGems Python and C++ callers, and the TileOps stubs, all registered against `PrivateUse1`. -- `csrc/aten/common.cc` reads the routing table at first dispatch: the table PyTorch-Plugin-FL selected for its build, or the file named by `FLAGOS_BACKEND_CONFIG`. +- `csrc/aten/common.cc` reads the routing table at first dispatch: the table Torch-FL selected for its build, or the file named by `FLAGOS_BACKEND_CONFIG`. - Vendor-native kernels live under `csrc/aten/backends//` and are generated for the operator surface that vendor library actually exports; operations the generator cannot match are skipped with a warning and fall through to routing or CPU fallback. - Unrouted operators reach `cpu_fallback`, which runs the CPU reference implementation and copies the result back to the device. diff --git a/docs/pytorch_plugin_fl_en/overview/features.md b/docs/torch_fl_en/overview/features.md similarity index 97% rename from docs/pytorch_plugin_fl_en/overview/features.md rename to docs/torch_fl_en/overview/features.md index b6d8d15c29..4df8db3669 100644 --- a/docs/pytorch_plugin_fl_en/overview/features.md +++ b/docs/torch_fl_en/overview/features.md @@ -29,7 +29,7 @@ Current routing is always queryable: `torch_fl.backend_config_path()` reports th ## Execution paths -PyTorch-Plugin-FL implements four operator execution strategies, and a platform may combine several of them. These are implementation strategies, not user-selectable product tiers. +Torch-FL implements four operator execution strategies, and a platform may combine several of them. These are implementation strategies, not user-selectable product tiers. - **Native vendor kernels** — direct calls into the vendor runtime and operator libraries (ACLNN for Ascend, topsaten for Enflame GCU, mudnn for Moore Threads MUSA). The plugin generates bindings to each vendor's C/C++ API. - **Compatibility boxing** — zero-copy metadata conversion into an independent PyTorch dispatch key when the vendor stack exposes one that can coexist with `PrivateUse1`. CUDA boxing reuses NVIDIA kernels through an external `libtorch_cuda.so`; MetaX, PPU and Hygon DCU box their vendor torch builds the same way. diff --git a/docs/pytorch_plugin_fl_en/overview/overview.md b/docs/torch_fl_en/overview/overview.md similarity index 79% rename from docs/pytorch_plugin_fl_en/overview/overview.md rename to docs/torch_fl_en/overview/overview.md index b2ff1648a6..188d4313ab 100644 --- a/docs/pytorch_plugin_fl_en/overview/overview.md +++ b/docs/torch_fl_en/overview/overview.md @@ -1,12 +1,12 @@ -# PyTorch-Plugin-FL Overview +# Torch-FL Overview `torch_fl` is a custom PyTorch device plugin built on the `PrivateUse1` extension mechanism. It registers [FlagGems](https://github.com/flagos-ai/FlagGems) high-performance Triton operators, vendor-native operator libraries, and CUDA compatibility kernels behind one device name: `flagos`. -Accelerator vendors ship different runtimes, compiler stacks, and PyTorch integration strategies. PyTorch-Plugin-FL hides those differences behind a unified runtime and operator-routing layer, so users program against standard PyTorch APIs and a single device name, and the plugin selects a kernel implementation per operator based on platform capability and configuration. +Accelerator vendors ship different runtimes, compiler stacks, and PyTorch integration strategies. Torch-FL hides those differences behind a unified runtime and operator-routing layer, so users program against standard PyTorch APIs and a single device name, and the plugin selects a kernel implementation per operator based on platform capability and configuration. ## Design principles -PyTorch-Plugin-FL is built on five principles: +Torch-FL is built on five principles: 1. **PyTorch-native interface** — Standard PyTorch APIs work unchanged; users target the `flagos` device instead of vendor-specific extensions. 2. **One logical device** — A single device name (`flagos`) abstracts vendor differences. Platform-specific routing happens transparently at the operator level. @@ -41,7 +41,7 @@ Operator routing (FlagGems compiler kernels, vendor-native kernels, compatibilit | Experimental | Validation exists for a specific setup, model, or hardware environment; interfaces or build procedures may change. | | Runtime only | Device runtime support exists, but the platform is not a general eager operator backend. | -A capability existing in the PyTorch-Plugin-FL codebase does not imply that every platform implements or validates it. See the {doc}`compatibility matrix <../reference/compatibility>` for per-platform detail. +A capability existing in the Torch-FL codebase does not imply that every platform implements or validates it. See the {doc}`compatibility matrix <../reference/compatibility>` for per-platform detail. ```{toctree} :maxdepth: 2 diff --git a/docs/pytorch_plugin_fl_en/reference/compatibility.md b/docs/torch_fl_en/reference/compatibility.md similarity index 94% rename from docs/pytorch_plugin_fl_en/reference/compatibility.md rename to docs/torch_fl_en/reference/compatibility.md index 60aecd37df..990bcc6181 100644 --- a/docs/pytorch_plugin_fl_en/reference/compatibility.md +++ b/docs/torch_fl_en/reference/compatibility.md @@ -20,7 +20,7 @@ ### ATen minor-line pinning -PyTorch-Plugin-FL generates native bindings to PyTorch's internal ATen operator registry. Those bindings are sensitive to C++ ABI and operator schema changes, so the project pins to a PyTorch minor line — currently **2.10.x**. A different minor version (for example 2.11.x) produces build or runtime failures; patch releases inside the same line (2.10.0 → 2.10.1) are compatible. +Torch-FL generates native bindings to PyTorch's internal ATen operator registry. Those bindings are sensitive to C++ ABI and operator schema changes, so the project pins to a PyTorch minor line — currently **2.10.x**. A different minor version (for example 2.11.x) produces build or runtime failures; patch releases inside the same line (2.10.0 → 2.10.1) are compatible. ### Wheel compatibility record diff --git a/docs/pytorch_plugin_fl_en/reference/environment-variables.md b/docs/torch_fl_en/reference/environment-variables.md similarity index 95% rename from docs/pytorch_plugin_fl_en/reference/environment-variables.md rename to docs/torch_fl_en/reference/environment-variables.md index f6b4be6329..deff42c925 100644 --- a/docs/pytorch_plugin_fl_en/reference/environment-variables.md +++ b/docs/torch_fl_en/reference/environment-variables.md @@ -1,6 +1,6 @@ # Environment Variables -PyTorch-Plugin-FL has one namespace of its own, `FLAGOS_*`, plus a second set belonging to other projects (torch, FlagGems, FlagCX, TileLang, vendor SDKs) that it reads but does not own. This page documents the variables a user configures; the authoritative, complete list is the `VARIABLES` registry in `torch_fl/_env.py`, checked against the upstream `docs/reference/environment-variables.md` by a unit test. +Torch-FL has one namespace of its own, `FLAGOS_*`, plus a second set belonging to other projects (torch, FlagGems, FlagCX, TileLang, vendor SDKs) that it reads but does not own. This page documents the variables a user configures; the authoritative, complete list is the `VARIABLES` registry in `torch_fl/_env.py`, checked against the upstream `docs/reference/environment-variables.md` by a unit test. Nothing here is required to run a wheel: a wheel routes, compiles and runs with an empty environment. These variables select a different build, override a setting for measurement, or turn on a diagnostic. diff --git a/docs/pytorch_plugin_fl_en/release_notes/release-notes.md b/docs/torch_fl_en/release_notes/release-notes.md similarity index 90% rename from docs/pytorch_plugin_fl_en/release_notes/release-notes.md rename to docs/torch_fl_en/release_notes/release-notes.md index 930f948e9a..1a79c5e5b9 100644 --- a/docs/pytorch_plugin_fl_en/release_notes/release-notes.md +++ b/docs/torch_fl_en/release_notes/release-notes.md @@ -1,10 +1,10 @@ # Release Notes -This section includes the release information for PyTorch-Plugin-FL. +This section includes the release information for Torch-FL. ## v0.1.0 -Initial release of PyTorch-Plugin-FL as part of FlagOS. +Initial release of Torch-FL as part of FlagOS. - **One `flagos` device** — a `PrivateUse1`-based PyTorch device plugin; standard PyTorch APIs, tensor methods and storage work unchanged. - **Per-operator backend routing** — FlagGems Triton kernels, vendor-native operator libraries, CUDA compatibility boxing and explicit CPU fallback behind one device name, with a per-platform routing table and per-operator overrides. diff --git a/docs/pytorch_plugin_fl_zh/architecture/distributed.md b/docs/torch_fl_zh/architecture/distributed.md similarity index 91% rename from docs/pytorch_plugin_fl_zh/architecture/distributed.md rename to docs/torch_fl_zh/architecture/distributed.md index f6b20d8f19..e2a728302f 100644 --- a/docs/pytorch_plugin_fl_zh/architecture/distributed.md +++ b/docs/torch_fl_zh/architecture/distributed.md @@ -1,6 +1,6 @@ # 分布式集合通信 -PyTorch-Plugin-FL 通过 `ProcessGroupFlagOS` 为 `flagos` 设备提供分布式支持。它是原生的 `torch.distributed.ProcessGroup` 子类,在导入时完成注册,因此 `torch.distributed.init_process_group("flagos")` 可直接使用,无需对 `torch.distributed.*` 做任何 monkeypatch。 +Torch-FL 通过 `ProcessGroupFlagOS` 为 `flagos` 设备提供分布式支持。它是原生的 `torch.distributed.ProcessGroup` 子类,在导入时完成注册,因此 `torch.distributed.init_process_group("flagos")` 可直接使用,无需对 `torch.distributed.*` 做任何 monkeypatch。 ## 工作原理 @@ -45,7 +45,7 @@ flagos_dist.move_buffers_to_device(model, "flagos:0") ### DDP -导入时,PyTorch-Plugin-FL 会补丁 `torch.nn.parallel.DistributedDataParallel.__init__`。当模型位于 `flagos` 设备上时,该补丁会: +导入时,Torch-FL 会补丁 `torch.nn.parallel.DistributedDataParallel.__init__`。当模型位于 `flagos` 设备上时,该补丁会: - 强制使用 Python reducer,绕过 C++ reducer 的 CUDA 断言; - 替换默认的梯度累积钩子(其使用在 `privateuseone` 上没有分发的函数式集合通信),改为通过 `dist.all_reduce` 走 `ProcessGroupFlagOS`。 diff --git a/docs/pytorch_plugin_fl_zh/architecture/profiler.md b/docs/torch_fl_zh/architecture/profiler.md similarity index 96% rename from docs/pytorch_plugin_fl_zh/architecture/profiler.md rename to docs/torch_fl_zh/architecture/profiler.md index f55f441b2f..eb88dcd077 100644 --- a/docs/pytorch_plugin_fl_zh/architecture/profiler.md +++ b/docs/torch_fl_zh/architecture/profiler.md @@ -64,7 +64,7 @@ | 6 | `test_runtime_names_come_from_cbid` | 运行期事件名由 callback id 解码得到 | | 7 | `test_capture_window_containment` | 没有设备或运行期事件逸出采集时间窗 | -基线位于 `tests/data/profiler_cuda_baseline.json`。第 2 项断言刻意比上游 `torch.cuda` 更严格:它是 PyTorch-Plugin-FL 自身的约束,保留它是因为流向箭头出现悬空半边正是它要防的回归。 +基线位于 `tests/data/profiler_cuda_baseline.json`。第 2 项断言刻意比上游 `torch.cuda` 更严格:它是 Torch-FL 自身的约束,保留它是因为流向箭头出现悬空半边正是它要防的回归。 ## 已知缺口 diff --git a/docs/pytorch_plugin_fl_zh/architecture/torch-compile.md b/docs/torch_fl_zh/architecture/torch-compile.md similarity index 98% rename from docs/pytorch_plugin_fl_zh/architecture/torch-compile.md rename to docs/torch_fl_zh/architecture/torch-compile.md index 38b8be27b7..65366e9894 100644 --- a/docs/pytorch_plugin_fl_zh/architecture/torch-compile.md +++ b/docs/torch_fl_zh/architecture/torch-compile.md @@ -36,7 +36,7 @@ model = torch.compile(model, backend="flagos", options={"max_autotune": True}) - 它的 wheel 名为 `flagtree`,但安装的模块是 `triton`。 - 安装它会卸载官方 `triton` 并取而代之。 -- 因此 Inductor 自身的 `import triton` 在安装后就已经指向 FlagTree,PyTorch-Plugin-FL 不需要对 `sys.modules` 做任何补丁。 +- 因此 Inductor 自身的 `import triton` 在安装后就已经指向 FlagTree,Torch-FL 不需要对 `sys.modules` 做任何补丁。 后端编译器在 FlagTree 构建期通过 `FLAGTREE_BACKEND` 选择(NVIDIA 与 AMD 不设置),而不是运行期:同一份 Triton 内核代码为不同厂商后端编译。`is_flagtree_active()` 用于检测 FlagTree 构建,`FLAGOS_USE_FLAGTREE=1` 断言当前 Triton 必须是 FlagTree —— 不满足时直接报错,而不是静默使用官方 Triton 编译。 diff --git a/docs/pytorch_plugin_fl_zh/assets/images/pytorch-plugin-fl.png b/docs/torch_fl_zh/assets/images/torch-fl.png similarity index 100% rename from docs/pytorch_plugin_fl_zh/assets/images/pytorch-plugin-fl.png rename to docs/torch_fl_zh/assets/images/torch-fl.png diff --git a/docs/pytorch_plugin_fl_zh/getting_started/installation.md b/docs/torch_fl_zh/getting_started/installation.md similarity index 99% rename from docs/pytorch_plugin_fl_zh/getting_started/installation.md rename to docs/torch_fl_zh/getting_started/installation.md index 7895d4b7e3..0141c7e4ef 100644 --- a/docs/pytorch_plugin_fl_zh/getting_started/installation.md +++ b/docs/torch_fl_zh/getting_started/installation.md @@ -128,7 +128,7 @@ pip install torch==2.10.0 --index-url https://download.pytorch.org/whl/cpu FLAGOS_ACCELERATOR=gcu pip install --no-build-isolation -v -e . ``` -需要 TopsRider SDK(`libtopsrt.so` 运行时与 `libtopsaten.so` 算子库)。构建会运行 `scripts/codegen/codegen_gcu.py`,按已安装 `libtopsaten.so` 中实际存在的 `topsaten` 符号校验每个算子;SDK 中缺失的算子会给出告警并跳过。`torch-gcu` wheel 不能与 PyTorch-Plugin-FL 同时使用,因为它会自行占用 `PrivateUse1`。 +需要 TopsRider SDK(`libtopsrt.so` 运行时与 `libtopsaten.so` 算子库)。构建会运行 `scripts/codegen/codegen_gcu.py`,按已安装 `libtopsaten.so` 中实际存在的 `topsaten` 符号校验每个算子;SDK 中缺失的算子会给出告警并跳过。`torch-gcu` wheel 不能与 Torch-FL 同时使用,因为它会自行占用 `PrivateUse1`。 ### 摩尔线程 MUSA diff --git a/docs/pytorch_plugin_fl_zh/index.md b/docs/torch_fl_zh/index.md similarity index 81% rename from docs/pytorch_plugin_fl_zh/index.md rename to docs/torch_fl_zh/index.md index e84ac6c2fe..342a7dde18 100644 --- a/docs/pytorch_plugin_fl_zh/index.md +++ b/docs/torch_fl_zh/index.md @@ -1,8 +1,8 @@ -# PyTorch-Plugin-FL +# Torch-FL -PyTorch-Plugin-FL(`torch_fl`)是面向 FlagOS 软件栈的 PyTorch 设备插件。它对外提供统一的 `flagos` 设备,并在可复用原生内核、可移植编译器内核、厂商原生实现和显式 CPU 回退之间路由算子,使同一份 PyTorch 程序无需修改即可运行在不同加速器上。 +Torch-FL(`torch_fl`)是面向 FlagOS 软件栈的 PyTorch 设备插件。它对外提供统一的 `flagos` 设备,并在可复用原生内核、可移植编译器内核、厂商原生实现和显式 CPU 回退之间路由算子,使同一份 PyTorch 程序无需修改即可运行在不同加速器上。 -![PyTorch-Plugin-FL 架构](assets/images/pytorch-plugin-fl.png) +![Torch-FL 架构](assets/images/torch-fl.png) ::::{grid} 1 2 2 3 :gutter: 1 1 1 2 @@ -11,7 +11,7 @@ PyTorch-Plugin-FL(`torch_fl`)是面向 FlagOS 软件栈的 PyTorch 设备插 :link: overview/overview :link-type: doc -PyTorch-Plugin-FL 是什么、设计理念、能力范围与组件架构。 +Torch-FL 是什么、设计理念、能力范围与组件架构。 +++ [了解更多 »](overview/overview.md) @@ -69,8 +69,6 @@ PyTorch-Plugin-FL 是什么、设计理念、能力范围与组件架构。 :::: -## 项目链接 - - **代码仓库**:[flagos-ai/Torch-FL](https://github.com/flagos-ai/Torch-FL) - **FlagGems**:[flagos-ai/FlagGems](https://github.com/flagos-ai/FlagGems) - **FlagTree**:[flagos-ai/FlagTree](https://github.com/flagos-ai/FlagTree) diff --git a/docs/pytorch_plugin_fl_zh/overview/architecture.md b/docs/torch_fl_zh/overview/architecture.md similarity index 94% rename from docs/pytorch_plugin_fl_zh/overview/architecture.md rename to docs/torch_fl_zh/overview/architecture.md index 44b80ecbe3..eb8e5ff291 100644 --- a/docs/pytorch_plugin_fl_zh/overview/architecture.md +++ b/docs/torch_fl_zh/overview/architecture.md @@ -1,8 +1,8 @@ # 架构 -PyTorch-Plugin-FL 注册一个 PyTorch 设备,并对到达该设备的每个算子进行路由。 +Torch-FL 注册一个 PyTorch 设备,并对到达该设备的每个算子进行路由。 -![PyTorch-Plugin-FL 架构](../assets/images/pytorch-plugin-fl.png) +![Torch-FL 架构](../assets/images/torch-fl.png) ```text PyTorch API @@ -18,7 +18,7 @@ FlagGems/编译器内核 | 兼容性 boxing | 厂商原生内核 | CPU 回退 ## 设备注册 -导入时,PyTorch-Plugin-FL 接管 `PrivateUse1` dispatch key,并以 `flagos` 名称发布: +导入时,Torch-FL 接管 `PrivateUse1` dispatch key,并以 `flagos` 名称发布: 1. `torch.utils.rename_privateuse1_backend("flagos")` 为 dispatch key 命名。 2. `torch._register_device_module("flagos", flagos)` 安装设备模块,使 `torch.flagos.*` 可用。 diff --git a/docs/pytorch_plugin_fl_zh/overview/features.md b/docs/torch_fl_zh/overview/features.md similarity index 98% rename from docs/pytorch_plugin_fl_zh/overview/features.md rename to docs/torch_fl_zh/overview/features.md index d6a9839f80..21ba4c719f 100644 --- a/docs/pytorch_plugin_fl_zh/overview/features.md +++ b/docs/torch_fl_zh/overview/features.md @@ -29,7 +29,7 @@ ## 执行路径 -PyTorch-Plugin-FL 实现四类算子执行策略,同一平台可以组合多种。这些属于内部实现策略,而不是用户选择的产品等级。 +Torch-FL 实现四类算子执行策略,同一平台可以组合多种。这些属于内部实现策略,而不是用户选择的产品等级。 - **厂商原生内核** — 直接调用厂商运行时和算子库(Ascend 的 ACLNN、燧原 GCU 的 topsaten、摩尔线程 MUSA 的 mudnn)。插件为各厂商 C/C++ API 生成绑定代码。 - **兼容性 boxing** — 当厂商栈提供可与 `PrivateUse1` 共存的独立 PyTorch dispatch key 时,以零拷贝方式转换张量元数据。CUDA boxing 通过外部 `libtorch_cuda.so` 复用 NVIDIA 内核;MetaX、PPU 和海光 DCU 以同样方式对各自的厂商 torch 构建进行 boxing。 diff --git a/docs/pytorch_plugin_fl_zh/overview/overview.md b/docs/torch_fl_zh/overview/overview.md similarity index 80% rename from docs/pytorch_plugin_fl_zh/overview/overview.md rename to docs/torch_fl_zh/overview/overview.md index 234ca016d7..edf068f459 100644 --- a/docs/pytorch_plugin_fl_zh/overview/overview.md +++ b/docs/torch_fl_zh/overview/overview.md @@ -1,12 +1,12 @@ -# PyTorch-Plugin-FL 概览 +# Torch-FL 概览 `torch_fl` 是基于 `PrivateUse1` 扩展机制的自定义 PyTorch 设备插件。它将 [FlagGems](https://github.com/flagos-ai/FlagGems) 高性能 Triton 算子、厂商原生算子库和 CUDA 兼容内核统一注册到同一个设备名称下:`flagos`。 -不同加速器厂商提供的运行时、编译器栈和 PyTorch 集成方式各不相同。PyTorch-Plugin-FL 通过统一的运行时和算子路由层屏蔽这些差异:用户只使用标准 PyTorch API 和单一设备名称,插件根据平台能力与配置为每个算子选择内核实现。 +不同加速器厂商提供的运行时、编译器栈和 PyTorch 集成方式各不相同。Torch-FL 通过统一的运行时和算子路由层屏蔽这些差异:用户只使用标准 PyTorch API 和单一设备名称,插件根据平台能力与配置为每个算子选择内核实现。 ## 设计理念 -PyTorch-Plugin-FL 遵循五项原则: +Torch-FL 遵循五项原则: 1. **PyTorch 原生接口** — 标准 PyTorch API 无需修改;用户面向 `flagos` 设备编程,而不是使用厂商专用扩展。 2. **统一逻辑设备** — 单一设备名称(`flagos`)抽象厂商差异,平台相关路由在算子层透明完成。 @@ -41,7 +41,7 @@ print(y.cpu()) | 实验性 | 已在特定配置、模型或硬件环境中完成验证;接口或构建流程仍可能变化。 | | 仅运行时 | 已提供设备运行时支持,但该平台不是通用 eager 算子后端。 | -某项功能存在于 PyTorch-Plugin-FL 代码库中,并不代表每个平台都已实现或验证该功能。分平台详情请参阅 {doc}`兼容性矩阵 <../reference/compatibility>`。 +某项功能存在于 Torch-FL 代码库中,并不代表每个平台都已实现或验证该功能。分平台详情请参阅 {doc}`兼容性矩阵 <../reference/compatibility>`。 ```{toctree} :maxdepth: 2 diff --git a/docs/pytorch_plugin_fl_zh/reference/compatibility.md b/docs/torch_fl_zh/reference/compatibility.md similarity index 94% rename from docs/pytorch_plugin_fl_zh/reference/compatibility.md rename to docs/torch_fl_zh/reference/compatibility.md index b0db2aee45..6954b6c41c 100644 --- a/docs/pytorch_plugin_fl_zh/reference/compatibility.md +++ b/docs/torch_fl_zh/reference/compatibility.md @@ -20,7 +20,7 @@ ### ATen 次版本线固定 -PyTorch-Plugin-FL 会为 PyTorch 内部的 ATen 算子注册表生成原生绑定。这些绑定对 C++ ABI 与算子 schema 变化敏感,因此项目固定在某个 PyTorch 次版本线上 —— 当前为 **2.10.x**。使用不同的次版本(例如 2.11.x)会导致构建或运行失败;同一次版本线内的补丁版本(2.10.0 → 2.10.1)互相兼容。 +Torch-FL 会为 PyTorch 内部的 ATen 算子注册表生成原生绑定。这些绑定对 C++ ABI 与算子 schema 变化敏感,因此项目固定在某个 PyTorch 次版本线上 —— 当前为 **2.10.x**。使用不同的次版本(例如 2.11.x)会导致构建或运行失败;同一次版本线内的补丁版本(2.10.0 → 2.10.1)互相兼容。 ### wheel 兼容性记录 diff --git a/docs/pytorch_plugin_fl_zh/reference/environment-variables.md b/docs/torch_fl_zh/reference/environment-variables.md similarity index 94% rename from docs/pytorch_plugin_fl_zh/reference/environment-variables.md rename to docs/torch_fl_zh/reference/environment-variables.md index 2ebda80f9a..fb70d9b013 100644 --- a/docs/pytorch_plugin_fl_zh/reference/environment-variables.md +++ b/docs/torch_fl_zh/reference/environment-variables.md @@ -1,6 +1,6 @@ # 环境变量 -PyTorch-Plugin-FL 有自己的一套命名空间 `FLAGOS_*`,另外还会读取属于其他项目(torch、FlagGems、FlagCX、TileLang、厂商 SDK)的一组变量,但不拥有它们。本页记录用户需要配置的变量;完整且权威的清单是 `torch_fl/_env.py` 中的 `VARIABLES` 注册表,上游仓库的 `docs/reference/environment-variables.md` 与它逐项对齐,并由单元测试校验。 +Torch-FL 有自己的一套命名空间 `FLAGOS_*`,另外还会读取属于其他项目(torch、FlagGems、FlagCX、TileLang、厂商 SDK)的一组变量,但不拥有它们。本页记录用户需要配置的变量;完整且权威的清单是 `torch_fl/_env.py` 中的 `VARIABLES` 注册表,上游仓库的 `docs/reference/environment-variables.md` 与它逐项对齐,并由单元测试校验。 这里没有任何一项是运行 wheel 所必需的:wheel 在空环境下即可完成路由、编译与运行。这些变量用于选择不同的构建、为测量而覆盖某项设置,或打开诊断。 diff --git a/docs/pytorch_plugin_fl_zh/release_notes/release-notes.md b/docs/torch_fl_zh/release_notes/release-notes.md similarity index 90% rename from docs/pytorch_plugin_fl_zh/release_notes/release-notes.md rename to docs/torch_fl_zh/release_notes/release-notes.md index 5fdacf0980..35688e045a 100644 --- a/docs/pytorch_plugin_fl_zh/release_notes/release-notes.md +++ b/docs/torch_fl_zh/release_notes/release-notes.md @@ -1,10 +1,10 @@ # 发布说明 -本节包含 PyTorch-Plugin-FL 的发布信息。 +本节包含 Torch-FL 的发布信息。 ## v0.1.0 -PyTorch-Plugin-FL 作为 FlagOS 一部分的初始版本。 +Torch-FL 作为 FlagOS 一部分的初始版本。 - **统一的 `flagos` 设备** — 基于 `PrivateUse1` 的 PyTorch 设备插件;标准 PyTorch API、张量方法与存储无需修改即可使用。 - **按算子路由后端** — 在同一设备名称下整合 FlagGems Triton 内核、厂商原生算子库、CUDA 兼容性 boxing 与显式 CPU 回退,并提供分平台路由表与逐算子覆盖。