From e7d063a24354e377252c74abf4deed800e0f33cc Mon Sep 17 00:00:00 2001 From: gongchensu Date: Thu, 27 Aug 2026 15:55:03 +0800 Subject: [PATCH] feat(hygon): add modern stack validation workflow Provide a reproducible DTK development image while keeping source, models, and build output on the host. Document the modern Infini stack build using the InfiniCore and InfiniLM refactor branches and the InfiniRT, InfiniOps, and InfiniCCL master branches. Add the external Hygon operator selection, integration audit, and commands for the shared 9G cases 1-13. --- README.md | 5 + images/hygon/Dockerfile | 58 ++++++ images/hygon/README.md | 355 ++++++++++++++++++++++++++++++++ images/hygon/infiniops_ops.json | 18 ++ images/hygon/xmake-cache.lua | 3 + 5 files changed, 439 insertions(+) create mode 100644 images/hygon/Dockerfile create mode 100644 images/hygon/README.md create mode 100644 images/hygon/infiniops_ops.json create mode 100644 images/hygon/xmake-cache.lua diff --git a/README.md b/README.md index 42b00dd..a58684a 100644 --- a/README.md +++ b/README.md @@ -20,12 +20,17 @@ helpers, GitHub Actions matrix converter, reusable workflow, and tests. │ ├── metax/ │ ├── moore/ │ ├── cambricon/ +│ ├── hygon/ │ └── ascend/ └── tests/ ``` Prerequisites: Docker, Python 3.10+, and `pip install pyyaml`. +Platform development images have separate reproduction guides. See the +[Hygon development image](images/hygon/README.md) for modern InfiniLM +integration and the 9G cases 1-13. + ## Configuration For repository CI, the caller repository owns the project config, usually at diff --git a/images/hygon/Dockerfile b/images/hygon/Dockerfile new file mode 100644 index 0000000..aabc098 --- /dev/null +++ b/images/hygon/Dockerfile @@ -0,0 +1,58 @@ +ARG BASE_IMAGE=harbor.sourcefind.cn:5443/dcu/admin/base/custom:sglang-deepseek-v4-dev-zkjh +FROM ${BASE_IMAGE} + +SHELL ["/bin/bash", "-o", "pipefail", "-c"] + +ARG HTTP_PROXY +ARG HTTPS_PROXY +ARG NO_PROXY +ARG http_proxy +ARG https_proxy +ARG no_proxy + +# Keep the DTK-enabled PyTorch supplied by the base image when project +# dependencies are installed later in a development container. +RUN torch_version="$(python3 -m pip show torch | awk '/^Version:/ {print $2}')" && \ + test -n "$torch_version" && \ + printf 'torch==%s\n' "$torch_version" > /etc/pip-constraints.txt +ENV PIP_CONSTRAINT=/etc/pip-constraints.txt + +RUN python3 -m pip install --no-cache-dir \ + janus==2.0.0 \ + libclang==18.1.1 \ + regex==2026.4.4 \ + ruff==0.15.7 \ + safetensors==0.8.0 \ + sentencepiece==0.2.1 \ + tokenizers==0.22.2 \ + transformers==5.15.0 \ + xxhash==3.6.0 + +ARG XMAKE_VERSION=v3.0.5 +ARG XMAKE_SHA256=4075ef1b4cba7f5eb5c55a4451761ef081086bb3142e7d4b130cf9c7484e68f0 +ARG XMAKE_DOWNLOAD_URL=https://github.com/xmake-io/xmake/releases/download/${XMAKE_VERSION}/xmake-bundle-${XMAKE_VERSION}.linux.x86_64 +ENV XMAKE_ROOT=y +RUN curl --retry 5 --retry-all-errors -fsSL \ + "${XMAKE_DOWNLOAD_URL}" -o /usr/local/bin/xmake && \ + printf '%s %s\n' "$XMAKE_SHA256" /usr/local/bin/xmake \ + | sha256sum --check - && \ + chmod +x /usr/local/bin/xmake + +# Pin the package recipe used by the validated environment and pre-populate +# the Xmake package cache required by the InfiniLM native extensions. +ARG XMAKE_REPO_URL=https://gitee.com/crapromer/xmake-repo.git +ARG XMAKE_REPO_COMMIT=674a71c5905870e6e7e162bf689903822bbbf16e +COPY images/hygon/xmake-cache.lua /opt/xmake-cache-project/xmake.lua +RUN git clone "${XMAKE_REPO_URL}" /opt/xmake-repo && \ + git -C /opt/xmake-repo checkout "${XMAKE_REPO_COMMIT}" && \ + xmake repo --clear --global && \ + xmake repo --add --global xmake-repo /opt/xmake-repo && \ + xmake require -y -P /opt/xmake-cache-project + +ENV DTK_ROOT=/opt/dtk +ENV HYGON_ARCH=gfx936 +ENV CUDA_HOME=/opt/dtk/cuda/cuda +ENV PATH=/opt/dtk/cuda/cuda/bin:/root/.local/bin:${PATH} +ENV LD_LIBRARY_PATH=/opt/dtk/cuda/cuda/targets/x86_64-linux/lib:${LD_LIBRARY_PATH} + +WORKDIR /workspace/run diff --git a/images/hygon/README.md b/images/hygon/README.md new file mode 100644 index 0000000..1caa4a1 --- /dev/null +++ b/images/hygon/README.md @@ -0,0 +1,355 @@ +# Hygon development image + +Validate InfiniLM with InfiniCore's InfiniRT, InfiniOps, and InfiniCCL components +on Hygon DCUs with DTK 26.04. Source, weights, and build output stay on the host. + +The default base image is +`harbor.sourcefind.cn:5443/dcu/admin/base/custom:sglang-deepseek-v4-dev-zkjh`. +The development image preserves its Hygon PyTorch and FlashAttention builds and +pins the additional Python, Xmake, and Xmake-recipe versions. + +## Prepare source checkouts + +Use existing checkouts under `$HOME/codes` (or set `SOURCE_ROOT` to their parent +directory). The build uses InfiniCore's submodules, not separate top-level +InfiniRT, InfiniOps, or InfiniCCL checkouts. + +| Checkout under `SOURCE_ROOT` | Ref | +| --- | --- | +| `InfiniCore` | `refactor/component-manifest` | +| `InfiniCore/submodules/InfiniRT` | `master` | +| `InfiniCore/submodules/InfiniOps` | `master` | +| `InfiniCore/submodules/InfiniCCL` | `master` | +| `InfiniLM` | `refactor/adopt-modern-infini-stack` | + +If these refs and Core's component gitlinks are already prepared, skip to +[Build and start](#build-and-start). Otherwise, run on the host: + +```bash +SOURCE_ROOT="${SOURCE_ROOT:-$HOME/codes}" + +for checkout in InfiniCore InfiniLM; do + repo="$SOURCE_ROOT/$checkout" + test -d "$repo/.git" || { + echo "Missing checkout: $repo" >&2 + exit 1 + } + test -z "$(git -C "$repo" status --porcelain)" || { + echo "Checkout is not clean: $repo" >&2 + exit 1 + } +done + +checkout_ref() { + local checkout="$1" + local repository="$2" + local ref="$3" + git -C "$checkout" fetch "$repository" "$ref" + git -C "$checkout" switch --detach FETCH_HEAD + git -C "$checkout" submodule update --init --recursive +} + +checkout_ref "$SOURCE_ROOT/InfiniCore" \ + https://github.com/InfiniTensor/InfiniCore.git \ + refactor/component-manifest +checkout_ref "$SOURCE_ROOT/InfiniLM" \ + https://github.com/InfiniTensor/InfiniLM.git \ + refactor/adopt-modern-infini-stack + +for component in InfiniRT InfiniOps InfiniCCL; do + checkout_ref "$SOURCE_ROOT/InfiniCore/submodules/$component" \ + "https://github.com/InfiniTensor/$component.git" master +done + +git -C "$SOURCE_ROOT/InfiniCore" add \ + submodules/InfiniRT submodules/InfiniOps submodules/InfiniCCL +git -C "$SOURCE_ROOT/InfiniCore" \ + -c user.name='Infini integration validation' \ + -c user.email='integration@localhost' \ + commit -m 'chore: pin Hygon integration revisions' +``` + +This leaves existing branch tips unchanged. The local Core commit pins the +component revisions; the build checks that their worktrees match the gitlinks. + +## Build and start + +On the host, use this CI checkout and build the current Dockerfile. An older +image with the same tag may lack dependencies such as `janus`. Start a new +container to use the rebuilt image. + +```bash +cd /path/to/ci +CI_ROOT="$PWD" +test -f "$CI_ROOT/images/hygon/infiniops_ops.json" + +docker build \ + -f "$CI_ROOT/images/hygon/Dockerfile" \ + -t infinitensor/infini-dev:hygon-dtk2604 \ + "$CI_ROOT" +``` + +Create an isolated output directory and start the container: + +```bash +SOURCE_ROOT="${SOURCE_ROOT:-$HOME/codes}" +RUN_ROOT="$(mktemp -d "$HOME/infini-hygon-run-XXXXXX")" +printf 'SOURCE_ROOT=%s\nRUN_ROOT=%s\n' "$SOURCE_ROOT" "$RUN_ROOT" + +docker run --rm -it \ + --name infini-hygon-dev \ + --network host \ + --ipc host \ + --device /dev/kfd \ + --device /dev/mkfd \ + --device /dev/dri \ + --group-add video \ + --group-add render \ + --ulimit memlock=-1:-1 \ + --ulimit stack=67108864:67108864 \ + -v /opt/hyhal:/opt/hyhal:ro \ + -v /data/node28:/data/node28:ro \ + -v /home_aclsylqidf:/home_aclsylqidf:ro \ + -v "${SOURCE_ROOT:?SOURCE_ROOT is not set}:/workspace/src" \ + -v "${RUN_ROOT:?RUN_ROOT is not set}:/workspace/run" \ + -v "${CI_ROOT:?CI_ROOT is not set}:/workspace/ci:ro" \ + -e SOURCE_ROOT=/workspace/src \ + -e RUN_ROOT=/workspace/run \ + -w /workspace/src/InfiniLM \ + infinitensor/infini-dev:hygon-dtk2604 \ + bash +``` + +TP8 cases require eight devices. Each new `RUN_ROOT` keeps build output and +logs on the host and separates them from previous builds. + +Run the remaining commands inside the container. Trust the mounted checkouts +so container root can read Git metadata owned by the host user: + +```bash +for repo in \ + InfiniLM InfiniCore \ + InfiniCore/submodules/InfiniRT \ + InfiniCore/submodules/InfiniOps \ + InfiniCore/submodules/InfiniCCL +do + git config --global --add safe.directory "$SOURCE_ROOT/$repo" +done +``` + +Verify the host driver, Hygon PyTorch, and Janus before building: + +```bash +hy-smi +python3 - <<'PY' +import janus +import torch + +print("torch:", torch.__version__) +print("device count:", torch.cuda.device_count()) +print("device 0:", torch.cuda.get_device_name(0)) +PY +``` + +## Build the stack + +The operator config selects native implementation 0, ATen implementation 8 +for `Fill`/`Tril`/`Triu`, and linked implementation 16 for FlashAttention. + +```bash +BUILD_ROOT="$RUN_ROOT/build" +LOG_ROOT="$RUN_ROOT/logs" +mkdir -p "$BUILD_ROOT" "$LOG_ROOT" +set -o pipefail + +cd "$SOURCE_ROOT/InfiniLM" +python3 scripts/build_infini_stack.py \ + --infinicore-root "$SOURCE_ROOT/InfiniCore" \ + --backend hygon \ + --hygon-arch gfx936 \ + --operator-config /workspace/ci/images/hygon/infiniops_ops.json \ + --build-root "$BUILD_ROOT/stack" \ + --jobs 16 \ + --test 2>&1 | tee "$LOG_ROOT/build-stack.log" +``` + +`--test` includes InfiniRT tests and a two-device InfiniCCL AllReduce smoke +test. Continue only after it succeeds: + +```bash +export INFINI_ROOT="$BUILD_ROOT/stack/prefix" +export CUDA_COMPAT_LIB="$CUDA_HOME/targets/x86_64-linux/lib" +export LD_LIBRARY_PATH="$INFINI_ROOT/lib:$CUDA_COMPAT_LIB:${LD_LIBRARY_PATH:-}" + +xmake f -y -c -o "$BUILD_ROOT/InfiniLM" -m release +python3 -m pip install -e . --no-build-isolation --no-deps \ + 2>&1 | tee "$LOG_ROOT/build-infinilm.log" +python3 examples/bench.py --help +``` + +`--no-deps` uses the image's dependencies; `--help` checks the benchmark imports +before loading model weights. Continue only after both commands succeed. + +## Audit the integration + +Define the logging helper and model paths once for the audit and cases 1-13: + +```bash +run_case() { + local case_id="$1" + shift + python3 examples/bench.py --device hygon "$@" 2>&1 \ + | tee "$LOG_ROOT/${case_id}.log" +} + +MODEL_8B=/data/node28/shared/models/9g_8b_thinking_llama +MODEL_70B=/home_aclsylqidf/shared/FM9G_80B_SFT_MHA +``` + +Run this short audit first. A failed command stops the block and leaves the +container shell open; proceed only when the whole block succeeds: + +```bash +( +set -e -o pipefail +export INFINI_OPS_TRACE_CALLS=1 +export INFINICORE_GRAPH_DEBUG=1 + +run_case flash-graph-audit \ + --model="$MODEL_8B" \ + --enable-paged-attn \ + --attn=flash-attn \ + --enable-graph \ + --batch-size=1 \ + --input-len=32 \ + --output-len=2 + +AUDIT_LOG="$LOG_ROOT/flash-graph-audit.log" +for operator in FlashAttnVarlenFunc FlashAttnWithKvcache; do + grep -q "\"operator_name\": \"$operator\".*\"implementation\": 16" \ + "$AUDIT_LOG" +done +grep -q '"operator_name": "ReshapeAndCacheFlash".*"implementation": 0' \ + "$AUDIT_LOG" +grep -m1 'Using InfiniRT C++ segmented graph runtime API' "$AUDIT_LOG" +if grep -q 'Falling back to eager execution' "$AUDIT_LOG"; then + echo 'Graph audit failed: eager fallback detected' >&2 + exit 1 +fi +grep '^\[INFINI_OPS_TRACE_CALLS\]' "$AUDIT_LOG" | sort -u + +ldd "$SOURCE_ROOT/InfiniLM/python/infinicore/lib/libinfinicore_runtime.so" \ + | grep -E 'libinfiniops|libinfiniccl|libinfinirt' +) +``` + +The checks require both linked FlashAttention providers, native KV-cache +reshaping, and the InfiniRT graph runtime without eager fallback. + +## Run cases 1-13 + +Cases 1-4 cover 8B FlashAttention and graph mode: + +```bash +case_number=1 +for batch_size in 1 4 16 64; do + run_case "case-${case_number}-8b-flash-graph" \ + --warmup \ + --model="$MODEL_8B" \ + --enable-paged-attn \ + --attn=flash-attn \ + --enable-graph \ + --input-len=32,256,4096 \ + --output-len=256,1024,2048,4096 \ + --batch-size="$batch_size" + case_number=$((case_number + 1)) +done +``` + +Cases 5-6 cover 8B paged graph and static attention: + +```bash +run_case case-5-8b-paged-graph \ + --model="$MODEL_8B" \ + --enable-paged-attn \ + --enable-graph \ + --batch-size=32 \ + --input-len=2048 \ + --output-len=2048 + +run_case case-6-8b-static \ + --model="$MODEL_8B" \ + --batch-size=4 \ + --input-len=1024 \ + --output-len=1024 +``` + +Cases 7-10 cover 70B FlashAttention and graph mode with TP8: + +```bash +case_number=7 +for batch_size in 1 4 16 64; do + run_case "case-${case_number}-70b-flash-graph" \ + --warmup \ + --model="$MODEL_70B" \ + --enable-paged-attn \ + --attn=flash-attn \ + --enable-graph \ + --input-len=32,256,4096 \ + --output-len=256,1024,2048,4096 \ + --batch-size="$batch_size" \ + --tp=8 + case_number=$((case_number + 1)) +done +``` + +Cases 11-13 cover MHA FlashAttention and the remaining 70B workloads. Case 13 +keeps the shared matrix command, which enables paged attention and graph mode +despite its historical `Static` label. + +```bash +run_case case-11-70b-mha-flash-graph \ + --model="$MODEL_70B" \ + --enable-paged-attn \ + --attn=flash-attn \ + --enable-graph \ + --pre-transpose \ + --tp=8 \ + --num-blocks=32 + +run_case case-12-70b-paged-graph \ + --model="$MODEL_70B" \ + --enable-paged-attn \ + --enable-graph \ + --batch-size=32 \ + --input-len=2048 \ + --output-len=2048 \ + --tp=8 + +run_case case-13-70b-paged-graph \ + --model="$MODEL_70B" \ + --enable-paged-attn \ + --enable-graph \ + --batch-size=4 \ + --input-len=1024 \ + --output-len=1024 \ + --tp=8 +``` + +## Reference Hygon results + +| Cases | Result | Notes | +| --- | --- | --- | +| 1-3 | Passed | 8B FlashAttention and graph, B1/B4/B16 | +| 4 | Capacity-limited | Completed 8/12 workloads before allocation failed | +| 5 | Passed | 8B paged attention and graph | +| 6 | Passed | 8B static attention | +| 7-8 | Passed | 70B FlashAttention and graph, TP8 | +| 9-10 | Capacity-limited | Model initialization or cache allocation exceeded memory | +| 11 | Passed | MHA 70B FlashAttention and graph, TP8 | +| 12 | Capacity-limited | B32, I2048/O2048 exceeded memory | +| 13 | Passed | 70B paged attention and graph, TP8 | + +Cases 4, 9, 10, and 12 retain the original workloads so capacity behavior is +visible instead of silently reducing their sizes. Re-run the audit and relevant +cases whenever one of the selected branch heads changes. diff --git a/images/hygon/infiniops_ops.json b/images/hygon/infiniops_ops.json new file mode 100644 index 0000000..7d0b70d --- /dev/null +++ b/images/hygon/infiniops_ops.json @@ -0,0 +1,18 @@ +{ + "add": {"implementations": [0]}, + "argmax": {"implementations": [0]}, + "copy": {"implementations": [0]}, + "embedding": {"implementations": [0]}, + "fill": {"implementations": [8]}, + "flash_attn_varlen_func": {"implementations": [16]}, + "flash_attn_with_kvcache": {"implementations": [16]}, + "fused_add_rms_norm": {"implementations": [0]}, + "gemm": {"implementations": [0]}, + "reshape_and_cache_flash": {"implementations": [0]}, + "rms_norm": {"implementations": [0]}, + "rotary_embedding": {"implementations": [0]}, + "silu_and_mul": {"implementations": [0]}, + "softmax": {"implementations": [0]}, + "tril": {"implementations": [8]}, + "triu": {"implementations": [8]} +} diff --git a/images/hygon/xmake-cache.lua b/images/hygon/xmake-cache.lua new file mode 100644 index 0000000..e4f4969 --- /dev/null +++ b/images/hygon/xmake-cache.lua @@ -0,0 +1,3 @@ +set_project("hygon-development-image-cache") + +add_requires("pybind11 v3.0.1")