Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
98 changes: 95 additions & 3 deletions .github/workflows/pr-test-xpu.yml
Original file line number Diff line number Diff line change
Expand Up @@ -28,7 +28,9 @@ jobs:
check-changes:
runs-on: ubuntu-latest
outputs:
main_package: ${{ steps.filter.outputs.main_package || steps.run-mode.outputs.run_all_tests }}
changes_exist: ${{ steps.filter.outputs.main_package == 'true' || steps.filter.outputs.multimodal_gen == 'true' || steps.run-mode.outputs.run_all_tests == 'true' }}
main_package: ${{ steps.filter.outputs.main_package == 'true' || steps.run-mode.outputs.run_all_tests == 'true' }}
multimodal_gen: ${{ steps.filter.outputs.multimodal_gen == 'true' || steps.run-mode.outputs.run_all_tests == 'true' }}
steps:
- name: Checkout code
uses: actions/checkout@v4
Expand Down Expand Up @@ -60,11 +62,16 @@ jobs:
- "python/sglang/kernels/aot/**/!(*.md|THIRDPARTYNOTICES.txt|LICENSE)"
- ".github/workflows/pr-test-xpu.yml"
- "docker/xpu.Dockerfile"
multimodal_gen:
- "python/sglang/multimodal_gen/**/!(*.md|*.ipynb)"
- "python/pyproject_xpu.toml"
- ".github/workflows/pr-test-xpu.yml"
- "docker/xpu.Dockerfile"

# ==================== PR Gate ==================== #
pr-gate:
needs: check-changes
if: needs.check-changes.outputs.main_package == 'true'
if: needs.check-changes.outputs.changes_exist == 'true'
uses: ./.github/workflows/pr-gate.yml
secrets: inherit

Expand Down Expand Up @@ -249,15 +256,96 @@ jobs:
docker rmi -f "${CI_SGLANG_XPU_IMAGE}" || true
fi

# ==================== Multimodal Gen ==================== #
multimodal-gen-test-1-gpu-xpu:
needs: [check-changes, pr-gate]
if: needs.check-changes.outputs.multimodal_gen == 'true'
runs-on: bmg-multigen-models
env:
DOCKERHUB_INTEL_USERNAME: ${{ secrets.DOCKERHUB_INTEL_USERNAME }}
DOCKERHUB_INTEL_TOKEN: ${{ secrets.DOCKERHUB_INTEL_TOKEN }}
steps:
- name: Reset workspace ownership
run: |
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
chown -R "$(id -u):$(id -g)" /w || true

- name: Checkout code
uses: actions/checkout@v4
with:
fetch-depth: 0
ref: ${{ inputs.ref || github.ref }}

- name: Start CI container
run: |
export HF_TOKEN="$(cat ~/huggingface_token.txt)"
bash scripts/ci/xpu/xpu_ci_start_container.sh
env:
GITHUB_WORKSPACE: ${{ github.workspace }}

- name: Install Dependency
timeout-minutes: 60
run: |
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --upgrade pip
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install pytest expecttest ray huggingface_hub tabulate "lmcache>=0.3.9"
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip uninstall -y flashinfer-python sgl-kernel sglang
docker exec ci_sglang_xpu cp /sglang-checkout/python/pyproject_xpu.toml /sglang-checkout/python/pyproject.toml
# Fetch tags so setuptools_scm resolves a real version instead of
# falling back to 0.0.0 on a shallow/tag-less checkout.
docker exec -w /sglang-checkout ci_sglang_xpu git fetch origin '+refs/tags/*:refs/tags/*' --force
docker exec -w /sglang-checkout/python ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir . --extra-index-url https://download.pytorch.org/whl/xpu
docker exec ci_sglang_xpu /opt/venv/bin/python3 -m pip install --no-cache-dir --no-deps xgrammar==0.1.33
docker exec ci_sglang_xpu /bin/bash -c '/opt/venv/bin/hf auth login --token ${HF_TOKEN}'

- name: Run diffusion server tests (1-GPU)
timeout-minutes: 60
run: |
docker exec ci_sglang_xpu bash -c "source /opt/venv/bin/activate && cd /sglang-checkout/python && python3 sglang/multimodal_gen/test/run_suite.py --suite 1-gpu-xpu"

- name: Cleanup container
if: always()
run: |
docker run --rm -v "${{ github.workspace }}:/w" busybox:latest \
chown -R "$(id -u):$(id -g)" /w || true
rm -rf \
python/build \
python/dist \
python/sglang.egg-info \
python/sglang/*.egg-info \
test/result.jsonl \
test/results \
test/.pytest_cache \
.pytest_cache || true
find . -type d -name "__pycache__" -prune -exec rm -rf {} + || true
find . -type f -name "*.pyc" -delete || true
# SIGTERM sglang and drain GPU context before `docker rm -f`;
# SIGKILL leaves the xe/GuC exec queue registered and triggers a
# GT reset (+ devcoredump) on B580.
if docker ps --format '{{.Names}}' | grep -qx ci_sglang_xpu; then
docker exec ci_sglang_xpu bash -c '
pkill -TERM -f "sglang|run_suite|python3.*test_" 2>/dev/null || true
for _ in $(seq 1 30); do
pgrep -f "sglang::|sglang.launch_server" >/dev/null || break
sleep 1
done
pkill -KILL -f "sglang|run_suite" 2>/dev/null || true
' || true
fi
docker rm -f ci_sglang_xpu || true
if [[ -n "${CI_SGLANG_XPU_IMAGE:-}" ]]; then
docker rmi -f "${CI_SGLANG_XPU_IMAGE}" || true
fi

finish:
if: always()
needs: [stage-a-test-1-gpu-xpu, stage-b-test-1-gpu-xpu, pr-gate]
needs: [stage-a-test-1-gpu-xpu, stage-b-test-1-gpu-xpu, multimodal-gen-test-1-gpu-xpu, pr-gate]
runs-on: ubuntu-latest
steps:
- name: Check job status
run: |
stage_a="${{ needs.stage-a-test-1-gpu-xpu.result }}"
stage_b="${{ needs.stage-b-test-1-gpu-xpu.result }}"
multimodal_gen="${{ needs.multimodal-gen-test-1-gpu-xpu.result }}"
if [ "$stage_a" != "success" ] && [ "$stage_a" != "skipped" ]; then
echo "stage-a failed with result: $stage_a"
exit 1
Expand All @@ -266,5 +354,9 @@ jobs:
echo "stage-b failed with result: $stage_b"
exit 1
fi
if [ "$multimodal_gen" != "success" ] && [ "$multimodal_gen" != "skipped" ]; then
echo "multimodal-gen failed with result: $multimodal_gen"
exit 1
fi
echo "All jobs completed successfully"
exit 0
Original file line number Diff line number Diff line change
Expand Up @@ -1223,8 +1223,11 @@ def _prepare_denoising_loop(self, batch: Req, server_args: ServerArgs):
image_kwargs = self.prepare_extra_func_kwargs(
getattr(self.transformer, "forward", self.transformer),
{
# Pass None (not []) so T2V paths whose transformer has no
# image_embedder skip the branch; diffusers guards on
# `is not None` only.
# TODO: make sure on-device
"encoder_hidden_states_image": image_embeds,
"encoder_hidden_states_image": image_embeds if image_embeds else None,
},
)

Expand Down
37 changes: 37 additions & 0 deletions python/sglang/multimodal_gen/test/server/gpu_cases.py
Original file line number Diff line number Diff line change
Expand Up @@ -1233,6 +1233,40 @@ def _make_5090_h3_consumer_budget_case() -> DiffusionTestCase:
ONE_GPU_5090_CASES.append(_make_5090_h3_consumer_budget_case())


# Intel Arc Pro B60 has 24 GiB of XPU memory, so only sub-~5B-parameter
# checkpoints fit fully resident. Larger cases in ONE_GPU_CASES (FLUX.1-dev,
# FLUX.2-dev, Qwen-Image, Hunyuan3D, SANA-Video, image-edit families) OOM on
# 24 GiB, and the FP8/NVFP4 quant paths are CUDA-only.
ONE_GPU_XPU_CASE_IDS = (
"zimage_image_t2i",
"flux_2_klein_image_t2i",
"flux_2_klein_base_image_t2i",
"wan2_1_t2v_1.3b",
)


def _select_xpu_cases(case_ids: tuple[str, ...]) -> list[DiffusionTestCase]:
cases_by_id = {case.id: case for case in ONE_GPU_CASES}
missing = [case_id for case_id in case_ids if case_id not in cases_by_id]
if missing:
raise RuntimeError(f"Unknown XPU diffusion case(s): {missing}")
return [cases_by_id[case_id] for case_id in case_ids]


# Consistency GT images are H100-generated; XPU output diverges at the
# pixel level (different attention kernels + fp reductions on Xe2) so
# SSIM/PSNR against the H100 golden always fails. test_server_1_gpu.py
# parametrizes directly from ONE_GPU_CASES, so mutate those entries in
# place -- overriding only via ONE_GPU_XPU_CASES would be ignored.
if current_platform.is_xpu():
_xpu_ids = set(ONE_GPU_XPU_CASE_IDS)
for _i, _case in enumerate(ONE_GPU_CASES):
if _case.id in _xpu_ids and _case.run_consistency_check:
ONE_GPU_CASES[_i] = replace(_case, run_consistency_check=False)

ONE_GPU_XPU_CASES = _select_xpu_cases(ONE_GPU_XPU_CASE_IDS)


# Nested unit/ tests verified to pass on AMD/ROCm as-is (no code change).
# Enabled incrementally and AMD-only: the CUDA `multimodal-gen-unit-test`
# lane keeps the flat glob below. Files that still need fixes/skips are added
Expand Down Expand Up @@ -1300,6 +1334,9 @@ def _discover_unit_tests() -> list[str]:
"1-gpu-5090": [
("test_server_1_gpu_5090.py", ONE_GPU_5090_CASES),
],
"1-gpu-xpu": [
("test_server_1_gpu.py", ONE_GPU_XPU_CASES),
],
"2-gpu": [
("test_server_2_gpu.py", TWO_GPU_CASES),
],
Expand Down
Loading
Loading