Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 10 additions & 4 deletions .github/script/sweep_common.py
Original file line number Diff line number Diff line change
Expand Up @@ -100,10 +100,16 @@ def build_parser(description: str) -> argparse.ArgumentParser:
parser.add_argument(
"--platform",
default="windows" if IS_WINDOWS else "linux",
# The -hrx variants are the same OS running an HRX build. Nothing in
# the sweep behaves differently; the label exists so the two builds'
# results do not land in the same filename or the same summary row.
choices=["linux", "windows", "linux-hrx", "windows-hrx"],
# Build variants use separate labels so their results do not land in
# the same filename or summary row.
choices=[
"linux",
"windows",
"linux-hrx",
"windows-hrx",
"linux-models-from-source",
"windows-models-from-source",
],
help="Label recorded in the output filename (default: autodetected).",
)
parser.add_argument("--host", default="127.0.0.1", help="Server bind address.")
Expand Down
12 changes: 12 additions & 0 deletions .github/script/sweep_summary.py
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,16 @@ def main(argv: list[str] | None = None) -> int:
parser.add_argument(
"--windows-hrx-result", default="", help="Job result for the Windows HRX sweep."
)
parser.add_argument(
"--linux-models-from-source-result",
default="",
help="Job result for the Linux models-from-source sweep.",
)
parser.add_argument(
"--windows-models-from-source-result",
default="",
help="Job result for the Windows models-from-source sweep.",
)
args = parser.parse_args(argv)

out: list[str] = ["# Model Sweep Results", ""]
Expand All @@ -46,6 +56,8 @@ def main(argv: list[str] | None = None) -> int:
("Linux (HRX)", args.linux_hrx_result),
("Windows", args.windows_result),
("Windows (HRX)", args.windows_hrx_result),
("Linux (models from source)", args.linux_models_from_source_result),
("Windows (models from source)", args.windows_models_from_source_result),
]
if any(result for _, result in overall):
out += ["| Platform | Overall |", "| --- | --- |"]
Expand Down
46 changes: 46 additions & 0 deletions .github/workflows/debian-portable.yml
Original file line number Diff line number Diff line change
Expand Up @@ -97,6 +97,52 @@ jobs:
retention-days: 7
if-no-files-found: error

- name: Build models from source
run: |
cd src
cmake --preset linux-portable \
-B build-models-from-source \
-DFLM_BUILD_GEMMA4E=ON
cmake --build build-models-from-source --parallel 2
- name: Create models-from-source package
env:
VERSION: ${{ steps.get_version.outputs.version }}
run: |
cd src/build-models-from-source
rm -rf install-models-from-source
DESTDIR=$PWD/install-models-from-source cmake --install .
package_root=install-models-from-source/opt/fastflowlm
mv "${package_root}/flm" "${package_root}/flm-real"
mv "${package_root}/flm-wrapper.sh" "${package_root}/flm"
chmod +x "${package_root}/flm"
strip --strip-debug "${package_root}/flm-real"
staged_engine="${package_root}/lib/libgemma4e_npu.so"
test -f "${staged_engine}"
if cmp -s "$GITHUB_WORKSPACE/src/lib/xrt/libgemma4e_npu.so" "${staged_engine}"; then
echo "The staged Gemma 4 engine matches the prebuilt library." >&2
exit 1
fi
readelf -n "${staged_engine}" | grep -A1 'Build ID'
tar czf "$GITHUB_WORKSPACE/models-from-source_fastflowlm_${VERSION}_linux.tar.gz" \
-C "${package_root}" \
flm \
flm-real \
lib/ \
xclbins/ \
model_list.json \
model_info.json
- name: Upload models-from-source build artifact
uses: actions/upload-artifact@v4
with:
name: models-from-source-portable
path: models-from-source_fastflowlm_${{ steps.get_version.outputs.version }}_linux.tar.gz
retention-days: 7
if-no-files-found: error

- name: Collect FFmpeg logs
if: always()
shell: bash
Expand Down
205 changes: 190 additions & 15 deletions .github/workflows/model-sweep.yml
Original file line number Diff line number Diff line change
Expand Up @@ -17,12 +17,10 @@ name: Model Sweep
# matrix job each, so a failure in one modality does not hide the others and
# "Re-run failed jobs" restarts only what actually broke.
#
# Each platform is swept twice: once against the default portable build and
# once against the HRX one, which is the same source built with FLM_USE_HRX=ON
# and a different set of op libraries underneath. Only the artifact differs, so
# the HRX jobs are their non-HRX counterparts with a different download step
# and a different --platform label. The HRX jobs are commit-sweep only, because
# releases do not currently ship an HRX asset.
# Each platform is swept against the default portable build and the HRX build.
# Commit sweeps also run Gemma 4 against separate Linux and Windows packages
# that build model engines from source. Releases do not ship these additional
# artifacts.

on:
pull_request:
Expand Down Expand Up @@ -92,11 +90,10 @@ jobs:
# what stops a sweep from silently picking up a stale artifact built from an
# earlier push on the same branch.
#
# Readiness is per build, not per workflow run: each of the four sweeps has
# its own <platform>[_hrx]_ready flag, decided by whether that one artifact
# exists. So an HRX build that fails costs you the HRX sweeps and nothing
# else, and vice versa. The build workflow already reports its own failure,
# so there is no reason to fail twice or to hold back a build that is fine.
# Readiness is per build, not per workflow run. Each artifact has its own
# readiness flag. A failed optional build does not block the other sweeps.
# The build workflow reports its failure, so this workflow skips the missing
# artifact.
resolve-build:
runs-on: ubuntu-latest
timeout-minutes: 90
Expand All @@ -107,7 +104,9 @@ jobs:
linux_run_id: ${{ steps.resolve.outputs.linux_run_id }}
windows_run_id: ${{ steps.resolve.outputs.windows_run_id }}
linux_ready: ${{ steps.resolve.outputs.linux_ready }}
linux_models_from_source_ready: ${{ steps.resolve.outputs.linux_models_from_source_ready }}
windows_ready: ${{ steps.resolve.outputs.windows_ready }}
windows_models_from_source_ready: ${{ steps.resolve.outputs.windows_models_from_source_ready }}
linux_hrx_ready: ${{ steps.resolve.outputs.linux_hrx_ready }}
windows_hrx_ready: ${{ steps.resolve.outputs.windows_hrx_ready }}

Expand Down Expand Up @@ -212,6 +211,7 @@ jobs:
wait_for_build() {
local workflow="$1" platform="$2" default_pattern="$3" hrx_pattern="$4"
local source_pattern="${5:-}" source_flag="${6:-}" source_label="${7:-}"
local run_id="" status artifacts
while :; do
Expand Down Expand Up @@ -254,14 +254,22 @@ jobs:
"${platform}_ready" "${platform} portable" "${run_id}"
check_artifact "${artifacts}" "${hrx_pattern}" \
"${platform}_hrx_ready" "${platform} HRX portable" "${run_id}"
if [ -n "${source_pattern}" ]; then
check_artifact "${artifacts}" "${source_pattern}" \
"${source_flag}" "${source_label}" "${run_id}"
fi
}
# Neither platform gates the other, so a failure here leaves that
# platform not-ready rather than aborting the step.
wait_for_build debian-portable.yml linux \
'^portable$' '^HRX-portable$' || echo "Linux build unavailable."
'^portable$' '^HRX-portable$' \
'^models-from-source-portable$' 'linux_models_from_source_ready' \
'Linux models-from-source portable' || echo "Linux build unavailable."
wait_for_build windows-build.yml windows \
'^fastflowlm-windows-[0-9a-f]+$' '^HRX-fastflowlm-windows-[0-9a-f]+$' \
'^models-from-source-fastflowlm-windows-[0-9a-f]+$' \
'windows_models_from_source_ready' 'Windows models-from-source portable' \
|| echo "Windows build unavailable."
sweep-linux:
Expand Down Expand Up @@ -352,7 +360,12 @@ jobs:
set -euo pipefail
python3 -m venv .venv
.venv/bin/python -m pip install --upgrade pip
.venv/bin/python -m pip install openai
for attempt in 1 2 3; do
.venv/bin/python -m pip install --no-cache-dir openai && break
if [ "${attempt}" -eq 3 ]; then
exit 1
fi
done
# Each script starts and stops its own flm server, picks its own models,
# and exits non-zero if any model failed.
Expand All @@ -378,6 +391,83 @@ jobs:
if-no-files-found: warn
retention-days: 30

sweep-linux-models-from-source:
needs: resolve-build
if: needs.resolve-build.outputs.linux_models_from_source_ready == 'true'
runs-on: [self-hosted, linux, x64, npu, sweep]
timeout-minutes: 60

strategy:
fail-fast: false
matrix:
task: [llm, vision]

env:
FLM_MODEL_PATH: /scratch

steps:
- name: Checkout sweep scripts
uses: actions/checkout@v4
with:
sparse-checkout: .github/script
filter: blob:none

- name: Download Linux models-from-source build
uses: dawidd6/action-download-artifact@v25
with:
github_token: ${{ secrets.GITHUB_TOKEN }}
run_id: ${{ needs.resolve-build.outputs.linux_run_id }}
name: models-from-source-portable
path: build-artifact
if_no_artifact_found: fail

- name: Unpack Linux models-from-source build
run: |
set -euo pipefail
mkdir -p flm-linux-models-from-source
tarball=$(ls build-artifact/models-from-source_fastflowlm_*_linux.tar.gz)
echo "Using tarball: ${tarball}"
tar xzf "${tarball}" -C flm-linux-models-from-source
chmod +x flm-linux-models-from-source/flm flm-linux-models-from-source/flm-real
echo "FLM_BIN=${GITHUB_WORKSPACE}/flm-linux-models-from-source/flm" >> "$GITHUB_ENV"
- name: Verify FLM binary runs
run: |
set -euo pipefail
"${FLM_BIN}" version
- name: Set up Python virtualenv
run: |
set -euo pipefail
python3 -m venv .venv
.venv/bin/python -m pip install --upgrade pip
for attempt in 1 2 3; do
.venv/bin/python -m pip install --no-cache-dir openai && break
if [ "${attempt}" -eq 3 ]; then
exit 1
fi
done
- name: Run Gemma 4 ${{ matrix.task }} sweep
run: |
set -euo pipefail
.venv/bin/python .github/script/sweep_${{ matrix.task }}.py \
--flm-bin "${FLM_BIN}" \
--platform linux-models-from-source \
--models gemma4-it:e2b gemma4-it:e4b \
--gen-lim "${GEN_LIM}" \
--request-timeout "${REQUEST_TIMEOUT}" \
--output-dir sweep-results
- name: Upload Gemma 4 ${{ matrix.task }} results
if: always()
uses: actions/upload-artifact@v4
with:
name: model-sweep-linux-models-from-source-${{ matrix.task }}-${{ github.run_number }}
path: sweep-results/
if-no-files-found: warn
retention-days: 30

# The Linux sweep again, against the HRX build. "Build Linux Portable"
# produces both from the same run, so this shares linux_run_id with
# sweep-linux but has its own readiness flag: whichever of the two builds
Expand Down Expand Up @@ -446,7 +536,12 @@ jobs:
set -euo pipefail
python3 -m venv .venv
.venv/bin/python -m pip install --upgrade pip
.venv/bin/python -m pip install openai
for attempt in 1 2 3; do
.venv/bin/python -m pip install --no-cache-dir openai && break
if [ "${attempt}" -eq 3 ]; then
exit 1
fi
done
# --platform is only a label, but it is the one that keeps these results
# apart from the default build's in the filenames and the summary table.
Expand Down Expand Up @@ -582,6 +677,84 @@ jobs:
if-no-files-found: warn
retention-days: 30

sweep-windows-models-from-source:
needs: resolve-build
if: needs.resolve-build.outputs.windows_models_from_source_ready == 'true'
runs-on: [self-hosted, windows, x64, npu, sweep]
timeout-minutes: 60

strategy:
fail-fast: false
matrix:
task: [llm, vision]

steps:
- name: Checkout sweep scripts
uses: actions/checkout@v4
with:
sparse-checkout: .github/script
filter: blob:none

- name: Resolve model path
shell: powershell
run: |
$modelPath = Join-Path $env:USERPROFILE ".flm"
New-Item -ItemType Directory -Force $modelPath | Out-Null
"FLM_MODEL_PATH=$modelPath" | Out-File -FilePath $env:GITHUB_ENV -Encoding utf8 -Append
- name: Download Windows models-from-source build
uses: dawidd6/action-download-artifact@v25
with:
github_token: ${{ secrets.GITHUB_TOKEN }}
run_id: ${{ needs.resolve-build.outputs.windows_run_id }}
name: models-from-source-fastflowlm-windows-${{ github.sha }}
path: flm-windows-models-from-source
if_no_artifact_found: fail

- name: Verify FLM binary runs
shell: powershell
run: |
$ErrorActionPreference = "Stop"
$flm = Join-Path $env:GITHUB_WORKSPACE "flm-windows-models-from-source\flm.exe"
if (-not (Test-Path $flm)) {
throw "flm.exe not found at $flm"
}
"FLM_BIN=$flm" | Out-File -FilePath $env:GITHUB_ENV -Encoding utf8 -Append
& $flm version
- name: Set up Python virtualenv
shell: powershell
run: |
$ErrorActionPreference = "Stop"
python -m venv .venv
.\.venv\Scripts\python.exe -m pip install --upgrade pip
.\.venv\Scripts\python.exe -m pip install openai
- name: Run models-from-source ${{ matrix.task }} sweep
shell: powershell
run: |
$ErrorActionPreference = "Stop"
$cmdArgs = @(
".github\script\sweep_${{ matrix.task }}.py",
"--flm-bin", $env:FLM_BIN,
"--platform", "windows-models-from-source",
"--models", "gemma4-it:e2b", "gemma4-it:e4b",
"--gen-lim", $env:GEN_LIM,
"--request-timeout", $env:REQUEST_TIMEOUT,
"--output-dir", "sweep-results"
)
& ".\.venv\Scripts\python.exe" $cmdArgs
if ($LASTEXITCODE -ne 0) { exit $LASTEXITCODE }
- name: Upload models-from-source ${{ matrix.task }} results
if: always()
uses: actions/upload-artifact@v4
with:
name: model-sweep-windows-models-from-source-${{ matrix.task }}-${{ github.run_number }}
path: sweep-results/
if-no-files-found: warn
retention-days: 30

# The Windows sweep again, against the HRX build. "Build Windows Packages"
# produces both from the same run, so this shares windows_run_id with
# sweep-windows but has its own readiness flag.
Expand Down Expand Up @@ -683,7 +856,7 @@ jobs:
# Reporting only, no NPU needed, so this one stays on a GitHub-hosted runner.
summary:
runs-on: ubuntu-latest
needs: [sweep-linux, sweep-windows, sweep-linux-hrx, sweep-windows-hrx]
needs: [sweep-linux, sweep-linux-models-from-source, sweep-windows, sweep-windows-models-from-source, sweep-linux-hrx, sweep-windows-hrx]
if: always()
steps:
# Only the sweep scripts and their test assets, never the sources: this
Expand Down Expand Up @@ -714,4 +887,6 @@ jobs:
--windows-result "${{ needs.sweep-windows.result }}" \
--linux-hrx-result "${{ needs.sweep-linux-hrx.result }}" \
--windows-hrx-result "${{ needs.sweep-windows-hrx.result }}" \
--linux-models-from-source-result "${{ needs.sweep-linux-models-from-source.result }}" \
--windows-models-from-source-result "${{ needs.sweep-windows-models-from-source.result }}" \
>> "$GITHUB_STEP_SUMMARY"
Loading
Loading