Repository navigation
632 lines (592 loc) · 27.8 KB
/
Copy pathpython.yaml
File metadata and controls
632 lines (592 loc) · 27.8 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
name: Python CI
on:
push:
branches:
- main
pull_request:
types: ['opened', 'reopened', 'synchronize', 'ready_for_review']
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: true
jobs:
lint:
# A push event cannot tell whether its branch belongs to a draft PR, so
# branch CI is driven by pull_request events and push is limited to main.
if: ${{ github.event_name != 'pull_request' || github.event.pull_request.draft == false }}
runs-on: ${{ matrix.os }}
strategy:
fail-fast: false
matrix:
os: [ "ubuntu-latest" ]
python-version: [ "3.10" ]
steps:
- name: Check out code
uses: actions/checkout@v7
with:
fetch-depth: 0
submodules: recursive
- name: Set up Python environment
uses: actions/setup-python@v4
with:
python-version: "3.10"
- name: Install pre-commit
run: pip install pre-commit
- name: Run pre-commit
run: pre-commit run --all-files
- name: Set up Node.js
uses: actions/setup-node@v1
with:
node-version: 20.19.0
# ESLint and Prettier must be in `package.json`
- name: Install Node.js dependencies
run: cd frontend && npm ci
- name: ESLint Check
run: cd frontend && npx eslint .
- name: Build frontend (static export)
run: cd frontend && npm run build
- name: Verify static export output
run: |
test -f frontend/out/index.html
test -f frontend/out/404.html
# Dynamic routes must emit __shell__ placeholder pages.
find frontend/out -name '__shell__*' | grep -q .
- name: Serve static export from backend
run: |
pip install fastapi httpx pytest xoscar
pytest --noconftest -vv \
xinference/api/tests/test_frontend_static.py \
xinference/api/tests/test_frontend_static_real_export.py
changes:
if: ${{ github.event_name != 'pull_request' || github.event.pull_request.draft == false }}
runs-on: ubuntu-latest
outputs:
gpu: ${{ steps.gpu.outputs.gpu }}
groups: ${{ steps.gpu.outputs.groups }}
steps:
- name: Check out code
uses: actions/checkout@v7
with:
fetch-depth: 0
submodules: recursive
- name: Detect GPU CI changes
id: gpu
shell: bash
run: |
set -uo pipefail
if [ "${{ github.event_name }}" = "pull_request" ]; then
base_sha="${{ github.event.pull_request.base.sha }}"
head_sha="${{ github.event.pull_request.head.sha }}"
# Match GitHub's PR diff: exclude commits added to the base branch
# after the PR branch diverged.
diff_range="$base_sha...$head_sha"
else
base_sha="${{ github.event.before }}"
head_sha="${{ github.sha }}"
diff_range="$base_sha..$head_sha"
fi
full=false
llm=false; embedding=false; image=false; audio=false
# Fall back to a full run when there is no usable diff base (new
# branch, or a force push whose previous head no longer exists).
if [ -z "$base_sha" ] || [[ "$base_sha" =~ ^0+$ ]] \
|| ! git diff --name-only "$diff_range" > changed-files.txt 2>/dev/null; then
full=true
else
# Map each changed path to the GPU test group(s) it affects.
# Paths matching no pattern (docs, examples, web UI, other
# workflows, ...) do not need GPU CI at all.
while IFS= read -r file; do
case "$file" in
xinference/model/audio/*) audio=true ;;
xinference/model/image/*) image=true ;;
xinference/model/embedding/*|xinference/model/rerank/*) embedding=true ;;
xinference/model/llm/*|xinference/model/scheduler/*|xinference/core/tests/test_continuous_batching.py) llm=true ;;
# Vendored model code is audio-only today (cosyvoice, f5_tts,
# matcha, fish_speech, ...).
xinference/thirdparty/*) audio=true ;;
# The web UI never affects GPU tests.
xinference/ui/*) ;;
# Tests under the shared runtime packages exercise CPU logic
# that the ubuntu/macos/windows CI already covers; editing a
# test file does not change the GPU launch/inference path, so
# it needs no GPU CI. (test_continuous_batching is matched by
# the llm rule above and still triggers the llm group.)
xinference/core/tests/*|xinference/api/tests/*|xinference/client/tests/*) ;;
# OAuth2 / auth and the admin & launch-history routers are pure
# HTTP/auth logic with no model inference path; the non-GPU CI
# covers them. Keep them out of GPU CI entirely.
xinference/api/oauth2/*|xinference/api/routers/admin.py|xinference/api/routers/launch_history.py) ;;
# Per-family HTTP routers map to a single GPU group instead of
# forcing a full run.
xinference/api/routers/llm.py) llm=true ;;
xinference/api/routers/embeddings.py|xinference/api/routers/rerank.py) embedding=true ;;
xinference/api/routers/images.py) image=true ;;
xinference/api/routers/audio.py) audio=true ;;
# Every GPU test launches models through a real RESTful API
# server subprocess and the xinference/client/ package (see
# conftest.py's `setup` fixture), so changes to either affect
# every family. The same fixture starts the test cluster through
# xinference/deploy/, and xinference/core/ is the actor runtime
# shared by every model family. Any of these affects every group.
# This also covers restful_api.py (the launch/inference
# endpoints), dependencies.py, and the models/videos routers.
xinference/core/*|xinference/api/*|xinference/client/*|xinference/deploy/*) full=true; break ;;
# video/ and flexible/ have no GPU test group at all: no
# group below runs a single test from either package, so a full
# run costs every GPU minute while covering none of the change.
# Their tests (video/tests/, flexible/tests/) run in the
# ubuntu/macos/windows CI, which is where they are covered.
xinference/model/video/*|xinference/model/flexible/*) ;;
# Remaining model families with no dedicated group, plus
# top-level files directly under xinference/model/ (batch.py,
# cache_manager.py, core.py, custom.py, utils.py) which are
# shared by every family; play it safe and run every group.
xinference/model/*) full=true; break ;;
# The rest of the package (deploy/, docs, examples, other
# top-level modules) is not exercised by any GPU test.
.github/workflows/python.yaml|.github/requirements/model-ci-constraints.txt|pyproject.toml|build_backend.py|build_web.py) full=true; break ;;
esac
done < changed-files.txt
fi
groups=""
if [ "$full" = true ]; then
groups="llm embedding image audio"
else
[ "$llm" = true ] && groups="$groups llm"
[ "$embedding" = true ] && groups="$groups embedding"
[ "$image" = true ] && groups="$groups image"
[ "$audio" = true ] && groups="$groups audio"
groups="${groups# }"
fi
gpu=true
[ -z "$groups" ] && gpu=false
echo "gpu=$gpu" >> "$GITHUB_OUTPUT"
echo "groups=$groups" >> "$GITHUB_OUTPUT"
echo "GPU CI required: $gpu (groups: ${groups:-none})"
build_test_job:
runs-on: ${{ matrix.os }}
needs: lint
env:
CONDA_ENV: test
defaults:
run:
shell: bash -l {0}
strategy:
fail-fast: false
matrix:
os: [ "ubuntu-latest", "macos-latest", "windows-latest" ]
python-version: [ "3.10", "3.11", "3.12", "3.13", "3.14" ]
module: [ "xinference" ]
exclude:
- { os: macos-latest, python-version: "3.11" }
- { os: macos-latest, python-version: "3.12" }
- { os: macos-latest, python-version: "3.13" }
- { os: windows-latest, python-version: "3.11" }
- { os: windows-latest, python-version: "3.12" }
- { os: windows-latest, python-version: "3.13" }
include:
- { os: macos-latest, module: metal, python-version: "3.10" }
steps:
- name: Check out code
uses: actions/checkout@v7
with:
fetch-depth: 0
submodules: recursive
- name: Set up conda ${{ matrix.python-version }}
uses: conda-incubator/setup-miniconda@v3
with:
python-version: ${{ matrix.python-version }}
activate-environment: ${{ env.CONDA_ENV }}
# Important for Python 3.12 and newer.
- name: Update pip and setuptools
if: ${{ matrix.python-version == '3.12' || matrix.python-version == '3.13' || matrix.python-version == '3.14' }}
run: |
python -m pip install -U pip "setuptools<82"
- name: Install PyTorch for Python 3.13 and newer
if: ${{ matrix.python-version == '3.13' || matrix.python-version == '3.14' }}
run: |
python -m pip install torch torchvision torchaudio
- name: Install NumPy 1.x where required
if: ${{ startsWith(matrix.os, 'windows') && matrix.python-version == '3.10' }}
run: |
python -m pip install "numpy<2"
- name: Install dependencies
env:
MODULE: ${{ matrix.module }}
OS: ${{ matrix.os }}
# Tests never touch the web UI; skip the npm build that the build
# backend would otherwise run on editable installs.
NO_WEB_UI: 1
run: |
if [ "$OS" == "ubuntu-latest" ]; then
sudo rm -rf /usr/share/dotnet
sudo rm -rf /opt/ghc
sudo rm -rf "/usr/local/share/boost"
sudo rm -rf "$AGENT_TOOLSDIRECTORY"
fi
if [ "$MODULE" != "metal" ]; then
export PIP_CONSTRAINT="$PWD/.github/requirements/model-ci-constraints.txt"
fi
pip install -e ".[dev]"
pip install "xllamacpp>=0.2.0"
if [ "$MODULE" == "metal" ]; then
conda install -c conda-forge "ffmpeg<7"
pip install "mlx>=0.22.0"
pip install mlx-lm
pip install "mlx-vlm>=0.3.4"
pip install mlx-whisper
# F5-TTS owns a model-specific MLX pin in its virtualenv recipe.
# Keep it out of the host env so the integration test exercises
# that isolated dependency set instead of inheriting this one.
pip install qwen-vl-utils!=0.0.9
pip install tomli
else
pip install transformers
pip install attrdict
pip install "timm>=0.9.16"
if [ "${{ matrix.python-version }}" != "3.13" ] && [ "${{ matrix.python-version }}" != "3.14" ]; then
pip install torch torchvision
fi
pip install accelerate
pip install sentencepiece
pip install transformers_stream_generator
pip install bitsandbytes
pip install "sentence-transformers>=5.1.1,<6"
pip install modelscope
pip install diffusers
pip install protobuf
pip install "FlagEmbedding==1.3.5"
pip install "tenacity>=8.2.0,<8.4.0"
pip install "jinja2==3.1.2"
pip install jj-pytorchvideo
pip install qwen-vl-utils!=0.0.9
pip install datamodel_code_generator
pip install jsonschema
pip install -r .github/requirements/model-ci-constraints.txt
python -c "from datasets import Dataset; from transformers import AutoModel; from peft import PeftModel; from sentence_transformers import SentenceTransformer, CrossEncoder; from FlagEmbedding import BGEM3FlagModel; from diffusers import DiffusionPipeline"
fi
working-directory: .
- name: Clean up disk
if: |
(startsWith(matrix.os, 'ubuntu'))
run: |
sudo rm -rf /usr/share/dotnet
sudo rm -rf /usr/local/lib/android
sudo rm -rf /opt/ghc
sudo apt-get clean
sudo rm -rf /var/lib/apt/lists/*
df -h
- name: Fix SSL on Windows
if: startsWith(matrix.os, 'windows')
shell: bash
run: |
echo "activate conda env"
source $CONDA/etc/profile.d/conda.sh || true
conda activate $CONDA_ENV || true
python -V
which python
echo "before: $SSL_CERT_FILE"
python -m pip install --quiet certifi || true
SSL_CERT_FILE=$(python -c "import certifi,os;print(os.path.normpath(certifi.where()))")
export SSL_CERT_FILE
export REQUESTS_CA_BUNDLE=$SSL_CERT_FILE
export CURL_CA_BUNDLE=$SSL_CERT_FILE
echo "after: $SSL_CERT_FILE"
echo "SSL_CERT_FILE=$(python -c 'import certifi;print(certifi.where())')" >> $GITHUB_ENV
- name: Test with pytest
env:
MODULE: ${{ matrix.module }}
PYTORCH_MPS_HIGH_WATERMARK_RATIO: 1.0
PYTORCH_MPS_LOW_WATERMARK_RATIO: 0.2
XFORMERS_FORCE_DISABLE_TRITON: 1
TORCH_DISABLE_FLASH_ATTENTION: 1
run: |
set -e
run_pytest() {
pytest --timeout=3000 \
-W ignore::PendingDeprecationWarning \
--cov-config=pyproject.toml \
--cov-report=xml \
--cov=xinference \
"$@"
}
if [ "$MODULE" == "metal" ]; then
status=0
mkdir -p test-results
python -m pip freeze > test-results/metal-host-packages.txt
for test_file in \
xinference/model/llm/mlx/tests/test_mlx.py \
xinference/model/llm/mlx/tests/test_xavier.py \
xinference/model/audio/tests/test_whisper_mlx.py \
xinference/model/audio/tests/test_f5tts_mlx.py \
xinference/model/llm/mlx/tests/test_distributed_model.py; do
group=$(basename "$test_file" .py)
echo "::group::Metal tests: $group"
# Keep separate processes for model environments, and run every
# group even if an earlier one fails. Preserve the failure status.
run_pytest --junitxml="test-results/$group.xml" "$test_file" || status=1
echo "::endgroup::"
done
[ "$status" -eq 0 ]
else
run_pytest \
-vv \
--ignore xinference/core/tests/test_continuous_batching.py \
--ignore xinference/model/image/tests/test_stable_diffusion.py \
--ignore xinference/model/image/tests/test_got_ocr2.py \
--ignore xinference/model/audio/tests \
--ignore xinference/model/embedding/tests/test_integrated_embedding.py \
--ignore xinference/model/llm/transformers/tests/test_tensorizer.py \
--ignore xinference/model/llm/tests/test_llm_model.py \
--ignore xinference/model/llm/vllm \
--ignore xinference/model/llm/sglang \
--ignore xinference/client/tests/test_client.py \
--ignore xinference/client/tests/test_async_client.py \
--ignore xinference/model/llm/mlx \
xinference
run_pytest \
benchmark/tests/test_benchmark_pd.py \
benchmark/tests/test_benchmark_sglang_pd.py \
benchmark/tests/test_analyze_pd_profile.py \
xinference/model/llm/sglang/tests/test_xavier.py \
xinference/model/llm/sglang/tests/test_xavier_pd.py \
xinference/model/llm/sglang/tests/test_pd.py \
xinference/model/llm/sglang/tests/test_xavier_gpu_lifecycle.py \
xinference/model/llm/sglang/tests/test_gc_lifecycle.py \
xinference/model/llm/sglang/tests/test_runtime.py \
xinference/model/llm/vllm/tests/test_distributed_executor_v1.py \
xinference/model/llm/vllm/tests/test_executor_routing.py \
xinference/model/llm/vllm/tests/test_speculative_config.py \
xinference/model/llm/vllm/xavier/test/test_profiling.py \
xinference/model/llm/vllm/xavier/test/test_actor_loop.py \
xinference/model/llm/vllm/tests/test_pd.py \
xinference/model/llm/vllm/xavier/test/test_block_tracker_reuse.py \
xinference/model/llm/vllm/xavier/test/test_remote_kvcache_manager.py \
xinference/model/llm/vllm/xavier/test/test_v1_engine.py \
xinference/model/llm/vllm/xavier/test/test_gpu_transfer.py \
xinference/model/llm/vllm/xavier/test/test_kv_schema.py \
xinference/model/llm/vllm/xavier/test/test_cross_engine.py \
xinference/model/llm/vllm/xavier/test/test_direct_handoff.py \
xinference/model/llm/vllm/xavier/test/test_direct_history.py \
xinference/model/llm/vllm/xavier/test/test_tiered_snapshot.py \
xinference/model/llm/vllm/xavier/test/test_request_transfer.py \
xinference/model/llm/vllm/xavier/test/test_transfer_blocks.py \
xinference/model/llm/vllm/xavier/test/test_review_regressions.py
fi
working-directory: .
- name: Upload Metal test results
if: ${{ always() && matrix.module == 'metal' }}
uses: actions/upload-artifact@v4
with:
name: metal-test-results-${{ matrix.python-version }}
path: test-results/
if-no-files-found: ignore
gpu_test_job:
name: build_test_job (gpu-t4, gpu, 3.12)
runs-on: gpu-t4
needs: [lint, changes]
# Runners are billed per-minute: skip when no GPU-relevant path changed,
# and only run for PRs and pushes to main (a push to a PR branch already
# triggers the pull_request run).
if: ${{ needs.changes.outputs.gpu == 'true' && (github.event_name == 'pull_request' || github.ref == 'refs/heads/main') }}
timeout-minutes: 90
env:
CONDA_ENV: test
# GPU tests never touch the web UI; skipping the npm build saves
# minutes. The API server only warns when the UI bundle is absent.
NO_WEB_UI: 1
defaults:
run:
shell: bash -l {0}
steps:
- name: Check out code
uses: actions/checkout@v7
with:
fetch-depth: 0
submodules: recursive
- name: Set up conda 3.12
uses: conda-incubator/setup-miniconda@v3
with:
python-version: "3.12"
activate-environment: ${{ env.CONDA_ENV }}
- name: Restore pip package cache
uses: actions/cache@v4
with:
path: ~/.cache/pip
key: gpu-pip-${{ hashFiles('.github/workflows/python.yaml', '.github/requirements/model-ci-constraints.txt', 'pyproject.toml') }}
restore-keys: gpu-pip-
# funasr downloads its auxiliary models (fsmn-vad, ct-punc, cam++) from
# ModelScope, whose CDN is very slow from GitHub's US runners (~15min
# per run). They are only ~1.5GB in total, so cache them. Main models
# come from HuggingFace (fast) and are intentionally not cached — all
# of them together would blow the 10GB per-repo cache limit.
# NOTE: xinference redirects MODELSCOPE_CACHE to ~/.xinference/modelscope
# (see get_xinference_home in xinference/constants.py), so that is the
# directory to cache — ~/.cache/modelscope is never created.
- name: Restore ModelScope model cache
uses: actions/cache@v4
with:
path: ~/.xinference/modelscope
key: gpu-modelscope-${{ hashFiles('xinference/model/audio/models/*.json') }}
restore-keys: gpu-modelscope-
- name: Update pip and setuptools
run: |
python -m pip install -U pip "setuptools<82"
- name: Install dependencies
run: |
export PIP_CONSTRAINT="$PWD/.github/requirements/model-ci-constraints.txt"
pip install -U -e ".[audio,dev]"
pip install -U "openai>1"
pip install -U modelscope
pip install -U gguf
pip install -U uv
pip install -U sse_starlette
pip install -U xoscar
pip install -U "python-jose[cryptography]"
pip install -U "passlib[bcrypt]"
pip install -U "aioprometheus[starlette]"
pip install -U "pynvml"
pip install transformers
pip install "funasr==1.2.7"
conda install -y -c conda-forge "pynini==2.1.5" "ffmpeg<7"
pip install -U "nemo_text_processing<1.1.0"
pip install -U omegaconf~=2.3.0
pip install -U "WeTextProcessing<1.0.4"
pip install -U librosa
pip install -U xxhash
pip install -U "ChatTTS==0.2.4"
pip install -U HyperPyYAML
pip uninstall -y matcha-tts
pip install -U 'onnxruntime-gpu==1.22.0; sys_platform == "linux"'
pip install -U openai-whisper
pip install -U "torch==2.7.0" "torchaudio==2.7.0" "torchvision==0.22.0"
pip install -U "loguru"
pip install -U "natsort"
pip install -U "loralib"
pip install -U "ormsgpack"
pip uninstall -y opencc
pip uninstall -y "faster_whisper"
pip install -U accelerate
pip install -U verovio
pip install -U cachetools
pip install -U silero-vad
pip install -U pydantic
pip install -U "diffusers==0.35.1"
pip install -U onnx
pip install -U onnxconverter_common
pip install -U torchdiffeq
pip install -U "x_transformers>=1.31.14"
pip install -U pypinyin
pip install -U tomli
pip install -U vocos
pip install -U jieba
pip install -U soundfile
pip install tensorizer
pip install -U "sentence-transformers==5.1.1"
pip install -U "FlagEmbedding==1.3.5"
pip install -U "peft>=0.18.0"
pip install "xllamacpp>=0.2.0" --index-url https://xorbitsai.github.io/xllamacpp/whl/cu124 --extra-index-url https://pypi.org/simple
pip install hf_transfer
# espeak-ng (TTS phonemization) is not packaged on conda-forge;
# install it from apt.
sudo apt-get update -qq
sudo apt-get install -y -qq espeak-ng
# Resolve the shared stack together, including transitive datasets.
pip install "numpy>=2" -r .github/requirements/model-ci-constraints.txt
python -c "from datasets import Dataset; from transformers import AutoModel; from peft import PeftModel; from sentence_transformers import SentenceTransformer, CrossEncoder; from FlagEmbedding import BGEM3FlagModel; from diffusers import DiffusionPipeline"
working-directory: .
- name: Test with pytest
env:
XFORMERS_FORCE_DISABLE_TRITON: 1
TORCH_DISABLE_FLASH_ATTENTION: 1
# Speed up model downloads from the HuggingFace hub; runners are
# ephemeral so every run downloads the main models from scratch.
HF_HUB_ENABLE_HF_TRANSFER: 1
# Pin the model source instead of relying on locale auto-detection;
# ModelScope is very slow from US-hosted runners.
XINFERENCE_MODEL_SRC: huggingface
run: |
# One pytest process per selected group: one cold start per family
# instead of one per file, while an import error or crash in one
# family cannot take down the others. No --cov: nothing in this
# repo consumes the coverage report. A failing group does not stop
# the remaining groups, so one run reports every failure.
#
# On a failure, retry only the failed tests once via --last-failed.
# test_continuous_batching hits an intermittent KV-cache batching
# bug (~1 in 4 runs, tracked separately) that is a same-process
# race, not environment flakiness; each pytest invocation still
# gets fresh model/server subprocesses from the `setup` fixture, so
# a rerun is not just re-hitting stale state.
status=0
run_pytest() {
local rc=0
pytest --timeout=3000 \
-W ignore::PendingDeprecationWarning \
"$@" || rc=$?
if [ "$rc" -ne 0 ]; then
echo "::warning::pytest exited with code $rc for group '$group'; retrying failed tests once"
rc=0
pytest --timeout=3000 \
-W ignore::PendingDeprecationWarning \
--last-failed \
"$@" || rc=$?
fi
echo "group '$group' pytest exit code: $rc"
if [ "$rc" -ne 0 ]; then
echo "::warning::pytest exited with code $rc for group '$group' after retry"
status=1
fi
}
for group in ${{ needs.changes.outputs.groups }}; do
echo "::group::GPU tests: $group"
case "$group" in
llm)
run_pytest \
xinference/core/tests/test_continuous_batching.py \
xinference/model/llm/tests/test_llm_model.py \
xinference/model/llm/transformers/tests/test_tensorizer.py
;;
embedding)
# test_integrated_embedding must run first: the other three
# tests launch models in per-model virtualenvs, which leak
# their site-packages into the process environment, and models
# launched afterwards in the same pytest process import
# packages from the wrong env (seen with both the
# Qwen3-VL-Embedding and Qwen3-VL-Reranker virtualenvs).
run_pytest \
xinference/model/embedding/tests/test_integrated_embedding.py \
xinference/model/embedding/tests/test_qwen3_vl_engine_params.py \
xinference/model/embedding/vllm/tests/test_vllm_embedding.py \
xinference/model/rerank/tests/test_qwen3_vl_reranker_virtualenv.py
;;
image)
# test_got_ocr2 is intentionally omitted: it self-skips on
# transformers >= 4.48.0 and this job pins 4.53.2, so it never
# runs a single case here.
run_pytest \
xinference/model/image/tests/test_stable_diffusion.py
;;
audio)
# test_cosyvoice is intentionally omitted: all of its cases
# are @pytest.mark.skip'd (diffusers on this image is
# incompatible), so it never runs anything here.
run_pytest \
xinference/model/audio/tests/test_whisper.py \
xinference/model/audio/tests/test_funasr.py \
xinference/model/audio/tests/test_chattts.py \
xinference/model/audio/tests/test_f5tts.py \
xinference/model/audio/tests/test_melotts.py \
xinference/model/audio/tests/test_kokoro.py \
xinference/model/audio/tests/test_fish_speech.py \
xinference/model/audio/tests/test_megatts.py
;;
esac
echo "::endgroup::"
done
echo "aggregate pytest status: $status"
# Finish with a plain test instead of an explicit `exit`: two runs
# with every pytest group green still ended in exit code 1 through
# the login shell's explicit-exit path on this runner image.
[ "$status" -eq 0 ]
working-directory: .