Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
17 changes: 15 additions & 2 deletions .github/workflows/ci.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -60,7 +60,12 @@ jobs:
enable-cache: true
python-version: ${{ matrix.python-version }}

- run: uv sync --frozen --group dev --extra hf --extra compose
# `dev-vllm26`, not `dev`: the dev groups are now named after the vLLM line
# they install, because the GPU harness keys a stale 0.19.x version assertion
# off the bare name `dev` (see the comment in pyproject.toml). Nothing here is
# vLLM-version-specific -- it only needs pytest plus the CPU deps -- but there
# is one dev group per line and this is the one the lockfile resolves.
- run: uv sync --frozen --group dev-vllm26 --extra hf --extra compose

- name: Run CPU tests
run: |
Expand All @@ -77,8 +82,16 @@ jobs:
-v -s --tb=short -x \
--cov=granite_switch --cov-report=xml

# COVERAGE REPORTING MUST NOT GATE A PR. `fail_ci_if_error: false` already
# declared that intent, but v4 can fail the step *before* it consults that
# flag -- it exits nonzero on a token/rate-limit problem of its own, which is
# what turns a fully green CPU suite red (the "Run CPU tests" step passes on
# both 3.11 and 3.12; only this step fails). `continue-on-error` enforces the
# intent at the workflow level, where the action cannot override it. v5 is the
# supported line; v4 is deprecated.
- name: Upload coverage
uses: codecov/codecov-action@v4
uses: codecov/codecov-action@v5
continue-on-error: true
with:
token: ${{ secrets.CODECOV_TOKEN }}
files: coverage.xml
Expand Down
11 changes: 10 additions & 1 deletion .github/workflows/gpu-tests.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -110,12 +110,21 @@ jobs:
# conflicting groups), so each gets its own cluster pod. Only the dev* groups
# used because the bare vllm26/vllm27 groups omit pytest. The label encodes
# the vLLM minor the leg installs (the GPU harness asserts it matches).
#
# `dep_group` MUST NOT be the bare `dev`. The harness looks this string up in a
# table of its own pairing each name with a job tag and a vLLM line it asserts;
# `dev`'s entry is frozen at 0.19.x ("dep group: dev (job tag v19, expect vllm
# 0.19.x)"). With `dev` on the 0.26 line since #130, this leg installed 0.26
# correctly and then died on that entry before pytest ran -- "FATAL expected
# vllm 0.19.x for group dev but got 0.26.0" -- red on every run since. A name
# the table does not know is not an error; it falls back to "expect vllm any.x"
# (which is why the dev-vllm27 leg was never affected). See pyproject.toml.
strategy:
fail-fast: false # a vllm26 failure must not hide the vllm27 result
matrix:
include:
- label: vllm26
dep_group: dev
dep_group: dev-vllm26
- label: vllm27
dep_group: dev-vllm27
runs-on: [self-hosted, gpu]
Expand Down
8 changes: 5 additions & 3 deletions CONTRIBUTING.md
Original file line number Diff line number Diff line change
Expand Up @@ -19,10 +19,12 @@ Or via pip: `pip install uv`
```bash
git clone https://github.com/<your-username>/granite-switch.git
cd granite-switch
uv sync --group dev
uv sync --group dev-vllm26
```
> **On macOS (or any machine without a CUDA GPU):** the `dev` group pulls in vLLM + CUDA,
> which have no macOS wheels, so `uv sync --group dev` fails. Install the CPU-only subset
> The dev groups are named after the vLLM line they install (`dev-vllm26`,
> `dev-vllm27`); there is no bare `dev` group. Pick the line you want to work against.
> **On macOS (or any machine without a CUDA GPU):** these groups pull in vLLM + CUDA,
> which have no macOS wheels, so `uv sync --group dev-vllm26` fails. Install the CPU-only subset
> instead — enough for unit, HF, and compose work:
> ```bash
> uv sync --frozen --no-default-groups --extra hf --extra compose
Expand Down
2 changes: 1 addition & 1 deletion docs/AUDIO.md
Original file line number Diff line number Diff line change
Expand Up @@ -30,7 +30,7 @@ with an `--asr-model` your installed transformers supports (for example
uv sync --extra vllm --extra audio # or --extra vllm27 --extra audio

# Development / running the test suite (the dev groups include audio already)
uv sync --group dev # vLLM 0.26.x
uv sync --group dev-vllm26 # vLLM 0.26.x
uv sync --group dev-vllm27 # vLLM 0.27.x
```

Expand Down
44 changes: 39 additions & 5 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -58,26 +58,60 @@ markers = [
[dependency-groups]
vllm26 = ["vllm>=0.26.0,<0.27.0"]
vllm27 = ["vllm>=0.27.0,<0.28.0"]
# NO DEV GROUP MAY BE NAMED A BARE `dev`. The GPU cluster harness looks the
# `--dep-group` name up in a table of its own that pairs each name with a job tag
# and the vLLM line it then ASSERTS the pod installed. Observed from run logs:
#
# dep group: dev (job tag v19, expect vllm 0.19.x)
# dep group: dev-vllm20 (job tag v20, expect vllm 0.20.x)
# dep group: dev-vllm27 (job tag dev-vllm27, expect vllm any.x)
#
# Note the third line: an UNKNOWN name is not an error -- it falls back to
# `tag = the name itself, expect any.x`, i.e. no assertion at all. Only names the
# table knows carry a version claim, and `dev`'s claim was frozen at 0.19.x.
#
# So once #130 moved `dev` onto the 0.26 line, the leg installed 0.26 correctly and
# then died on that stale entry, before pytest ever started:
#
# torch 2.11.0+cu130 vllm 0.26.0
# FATAL expected vllm 0.19.x for group dev but got 0.26.0
#
# while `dev-vllm27`, absent from the table, sailed through on `any.x`. Renaming to
# `dev-vllm26` therefore fixes the leg unconditionally -- it stops addressing the
# one stale entry and lands on the no-assertion fallback. It does NOT depend on the
# harness parsing "26" out of the name; it cannot, as the table is explicit.
#
# TRADE-OFF, AND THE REAL FOLLOW-UP: `any.x` means no leg asserts its vLLM line any
# more, so a venv that silently resolved the wrong line would no longer be caught
# here. The fix is for the harness table to gain real `dev-vllm26/27/30` entries;
# until it does, the per-line `vllm26`/`vllm27` bounds below are what pin the line.
#
# This is a rename, not a new group: `dev` is gone rather than aliased, so the name
# carrying the stale 0.19.x claim cannot be passed by accident. Adding a parallel
# `dev-vllm26` and keeping `dev` was tried and rejected -- an extra conflicting
# group multiplies uv's resolution-marker space (+3093 net lines of uv.lock for an
# identical 275-package resolution), whereas the rename is diff-balanced.
#
# `audio` is included so the audio tests can actually run: the ASR path needs
# vLLM's audio deps (av/soundfile/resampy) at runtime, and no group pulled them
# in before (integration tests failed with ModuleNotFoundError on a synced pod).
dev = ["pytest", "pytest-cov", { include-group = "vllm26" }, "granite-switch[hf,compose,audio]"]
dev-vllm26 = ["pytest", "pytest-cov", { include-group = "vllm26" }, "granite-switch[hf,compose,audio]"]
dev-vllm27 = ["pytest", "pytest-cov", { include-group = "vllm27" }, "granite-switch[hf,compose,audio]"]
test = ["pytest", "pytest-cov", "bitsandbytes", "optimum-quanto", { include-group = "dev" }]
test = ["pytest", "pytest-cov", "bitsandbytes", "optimum-quanto", { include-group = "dev-vllm26" }]

[tool.uv]
default-groups = ["vllm26"]
conflicts = [
# group-vs-group
[{ group = "vllm26" }, { group = "vllm27" }],
[{ group = "dev" }, { group = "vllm27" }],
[{ group = "dev" }, { group = "dev-vllm27" }],
[{ group = "dev-vllm26" }, { group = "vllm27" }],
[{ group = "dev-vllm26" }, { group = "dev-vllm27" }],
[{ group = "dev-vllm27" }, { group = "vllm26" }],
# group-vs-extra
[{ group = "vllm26" }, { extra = "vllm27" }],
[{ group = "vllm27" }, { extra = "vllm" }],
[{ group = "vllm27" }, { extra = "tutorials" }],
[{ group = "dev" }, { extra = "vllm27" }],
[{ group = "dev-vllm26" }, { extra = "vllm27" }],
[{ group = "dev-vllm27" }, { extra = "vllm" }],
[{ group = "dev-vllm27" }, { extra = "tutorials" }],
# extra-vs-extra
Expand Down
Loading
Loading