diff --git a/.github/scripts/lib/set_env_common.sh b/.github/scripts/lib/set_env_common.sh new file mode 100644 index 00000000..b7cc936b --- /dev/null +++ b/.github/scripts/lib/set_env_common.sh @@ -0,0 +1,163 @@ +# Copyright 2026 FlagOS Contributors +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +# Shared helpers for the set_env_*.sh platform provisioning scripts. +# +# Sourced (not executed) by each script after REPO_ROOT and the version pins are +# set. These five functions were copied verbatim into every script -- `pip_retry` +# in five slightly different spellings, `flag_gems_installed`/`install_flag_gems` +# in six copies -- so a fix had to be applied seven times and was easy to miss. +# +# The interpreter is taken from `${VENV_PYTHON:-python}`: every script but MetaX +# sets VENV_PYTHON to its job-local venv; MetaX runs the image's /opt/venv via +# PATH and falls through to `python`. `pip_retry` additionally honours +# PIP_RETRY_PYTHON (a per-call override, used by the CUDA script's several +# interpreters), PIP_RETRY_TIMEOUT (default 300s) and PIP_RETRY_NO_CACHE +# (default 1; the CUDA script turns it off to reuse its build cache). +# +# The scripts must keep `set -euo pipefail` in effect; every variable read here +# is either set by the caller or has a default. + +# pip install with retries for large wheels on unstable networks. +pip_retry() { + local python_exe="${PIP_RETRY_PYTHON:-${VENV_PYTHON:-python}}" + local timeout="${PIP_RETRY_TIMEOUT:-300}" + local -a extra=() + if [[ "${PIP_RETRY_NO_CACHE:-1}" == "1" ]]; then + extra+=(--no-cache-dir) + fi + local attempt=1 + while true; do + # Raise pip's own retry limit and timeout for large wheels on unstable + # networks: the FlagTree wheel is hundreds of MB, and pip's default timeout + # (15s) and retries (5) are not enough when the mirror link drops + # mid-download. + if "$python_exe" -m pip install --retries 10 --timeout "$timeout" "${extra[@]}" "$@"; then + return 0 + fi + if (( attempt >= 5 )); then + echo "::error::pip install failed after $attempt attempts: $*" + return 1 + fi + echo "::warning::pip install attempt $attempt failed; retrying: $*" + attempt=$((attempt + 1)) + sleep 10 + done +} + +# The FlagGems install is a VCS install, and pip reports the exit status of its +# last step -- the wheel build of whichever tree it managed to fetch. A checkout +# the runner's proxy truncated therefore still ends in `Successfully installed`, +# and pip never notices. On 2026-09-18 the Ascend runner's clone spent ten +# minutes printing +# fatal: unable to access 'https://github.com/flagos-ai/FlagGems.git/': +# Proxy CONNECT aborted +# (364 times), alongside `error: unable to read sha1 file of ...` and +# `error: invalid object 100644 2e574121... for +# '.github/workflows/rule-check.yaml'` for the blobs it never received, and then +# reported +# Successfully installed flaggems_setup-0.0.0 +# -- a 2.1 MB stub named after the build scaffolding rather than the project, +# where the same revision produced the 10 MB +# flag_gems-5.4.0rc2.post1+g437ba3938 on every other platform that ran that +# morning. Nothing failed until the integration suite took its first FlagGems +# route, four minutes later. +# +# So check the install instead of trusting pip's status, and reinstall when it +# is wrong: the conf routes this platform's operators to flagos_python, so an +# unusable flag_gems is not a state a provisioning script may leave behind. +# +# Three attempts at most, and only for an install pip called successful: a pip +# failure has already been retried five times by pip_retry, and repeating that +# spends the job's budget on a link that is down rather than on a bad checkout. +flag_gems_installed() { + "${VENV_PYTHON:-python}" - "${FLAGGEMS_REVISION:0:9}" <<'PY' +import importlib.metadata as metadata +import importlib.util +import sys + +try: + version = metadata.version("flag_gems") +except metadata.PackageNotFoundError: + raise SystemExit("flag_gems is not installed") + +# A distribution can be installed with no importable package behind it, which +# is what a truncated checkout produces. +if importlib.util.find_spec("flag_gems") is None: + raise SystemExit(f"flag_gems {version} has no importable package") + +print(f"flag_gems {version}") +if sys.argv[1] not in version: + # Not a failure: a revision given as a branch name, or a tarball without + # git metadata, lands on a version string that names neither. Only warn -- + # reinstalling cannot change how the version was written. + print( + f"::warning::flag_gems {version} does not name the requested revision " + f"{sys.argv[1]}; the checkout it was built from may be incomplete", + file=sys.stderr, + ) +PY +} + +# Install FlagGems from the pinned revision, retrying until it is importable. +install_flag_gems() { + local attempt=1 + while true; do + if ! pip_retry --no-deps "git+${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}"; then + echo "::error::could not install FlagGems from ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" + return 1 + fi + if flag_gems_installed; then + return 0 + fi + if (( attempt >= 3 )); then + echo "::error::no usable flag_gems after $attempt installs of ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" + return 1 + fi + echo "::warning::install attempt $attempt left no usable flag_gems; reinstalling" + attempt=$((attempt + 1)) + # A truncated tree installs under the build scaffolding's name, so both + # names have to go for the next attempt to be read as a fresh result. + "${VENV_PYTHON:-python}" -m pip uninstall -y flag_gems flaggems_setup >/dev/null 2>&1 || true + sleep 10 + done +} + +# Drop a vendor-torch root from a colon-separated path list, so the isolated +# venv does not inherit the image's vendor torch. +strip_vendor_paths() { + local value="${1:-}" + local entry + local -a entries=() + local -a kept=() + IFS=: read -ra entries <<< "$value" + for entry in "${entries[@]}"; do + [[ -z "$entry" ]] && continue + case "$entry" in + "$VENDOR_TORCH_ROOT"|"$VENDOR_TORCH_ROOT"/*) ;; + *) kept+=("$entry") ;; + esac + done + local joined="" + for entry in "${kept[@]}"; do + joined="${joined:+$joined:}$entry" + done + printf '%s' "$joined" +} + +# True when the venv interpreter exists and has pip. +venv_is_usable() { + [[ -x "$VENV_PYTHON" ]] || return 1 + "$VENV_PYTHON" -m pip --version >/dev/null 2>&1 +} diff --git a/.github/scripts/set_env_ascend.sh b/.github/scripts/set_env_ascend.sh index 1cf42e76..8555c44f 100755 --- a/.github/scripts/set_env_ascend.sh +++ b/.github/scripts/set_env_ascend.sh @@ -44,6 +44,10 @@ REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" # Shared version pins (torch, FlagTree, FlagGems); see .github/version-pins.env. # shellcheck source=.github/version-pins.env source "${REPO_ROOT}/.github/version-pins.env" + +# Shared set_env helpers (pip_retry, FlagGems install, path stripping). +# shellcheck source=.github/scripts/lib/set_env_common.sh +source "${REPO_ROOT}/.github/scripts/lib/set_env_common.sh" CPU_TORCH_INDEX_URL="${TORCH_FL_CPU_TORCH_INDEX_URL:-$CPU_TORCH_INDEX_URL_DEFAULT}" CPU_TORCH_VERSION="${TORCH_FL_CPU_TORCH_VERSION:-$CPU_TORCH_VERSION_DEFAULT}" # Default PyPI index for build deps (pip/setuptools/wheel/cmake/build/pytest). @@ -258,104 +262,7 @@ export PYTHONPATH="" # as the FlagGems-first Ascend conf dispatches an operator. # # Use --no-deps for the source packages so pip cannot replace the pinned CPU -# torch. Individual retries are intentional: the shared mirror can close a -# large-wheel response early, producing IncompleteRead even though the package -# is available. Retrying the failed package avoids restarting all setup. -pip_retry() { - local attempt=1 - while true; do - # Raise pip's own retry limit and timeout for large wheels on unstable - # networks: the flagtree wheel is ~200 MB, and the default timeout (15s) and - # retries (5) are not enough when the mirror link drops mid-download. - if "$VENV_PYTHON" -m pip install --retries 10 --timeout 300 --no-cache-dir "$@"; then - return 0 - fi - if (( attempt >= 5 )); then - echo "::error::pip install failed after $attempt attempts: $*" - return 1 - fi - echo "::warning::pip install attempt $attempt failed; retrying: $*" - attempt=$((attempt + 1)) - sleep 10 - done -} - -# The FlagGems install is a VCS install, and pip reports the exit status of its -# last step -- the wheel build of whichever tree it managed to fetch. A checkout -# the runner's proxy truncated therefore still ends in `Successfully installed`, -# and pip never notices. On 2026-09-18 the Ascend runner's clone spent ten -# minutes printing -# fatal: unable to access 'https://github.com/flagos-ai/FlagGems.git/': -# Proxy CONNECT aborted -# (364 times), alongside `error: unable to read sha1 file of ...` and -# `error: invalid object 100644 2e574121... for -# '.github/workflows/rule-check.yaml'` for the blobs it never received, and then -# reported -# Successfully installed flaggems_setup-0.0.0 -# -- a 2.1 MB stub named after the build scaffolding rather than the project, -# where the same revision produced the 10 MB -# flag_gems-5.4.0rc2.post1+g437ba3938 on every other platform that ran that -# morning. Nothing failed until the integration suite took its first FlagGems -# route, four minutes later. -# -# So check the install instead of trusting pip's status, and reinstall when it -# is wrong: the conf routes this platform's operators to flagos_python, so an -# unusable flag_gems is not a state this script may leave behind. -flag_gems_installed() { - "$VENV_PYTHON" - "${FLAGGEMS_REVISION:0:9}" <<'PY' -import importlib.metadata as metadata -import importlib.util -import sys - -try: - version = metadata.version("flag_gems") -except metadata.PackageNotFoundError: - raise SystemExit("flag_gems is not installed") - -# A distribution can be installed with no importable package behind it, which -# is what a truncated checkout produces. -if importlib.util.find_spec("flag_gems") is None: - raise SystemExit(f"flag_gems {version} has no importable package") - -print(f"flag_gems {version}") -if sys.argv[1] not in version: - # Not a failure: a revision given as a branch name, or a tarball without - # git metadata, lands on a version string that names neither. Only warn -- - # reinstalling cannot change how the version was written. - print( - f"::warning::flag_gems {version} does not name the requested revision " - f"{sys.argv[1]}; the checkout it was built from may be incomplete", - file=sys.stderr, - ) -PY -} - -# Three attempts at most, and only for an install pip called successful: a pip -# failure has already been retried five times by pip_retry, and repeating that -# spends the job's budget on a link that is down rather than on a bad checkout. -install_flag_gems() { - local attempt=1 - while true; do - if ! pip_retry --no-deps "git+${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}"; then - echo "::error::could not install FlagGems from ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" - return 1 - fi - if flag_gems_installed; then - return 0 - fi - if (( attempt >= 3 )); then - echo "::error::no usable flag_gems after $attempt installs of ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" - return 1 - fi - echo "::warning::install attempt $attempt left no usable flag_gems; reinstalling" - attempt=$((attempt + 1)) - # A truncated tree installs under the build scaffolding's name, so both - # names have to go for the next attempt to be read as a fresh result. - "$VENV_PYTHON" -m pip uninstall -y flag_gems flaggems_setup >/dev/null 2>&1 || true - sleep 10 - done -} - +# torch. # FlagTree installs the module named `triton`, so any stock or vendor Triton # already present would be shadowed rather than replaced, and the user manual # asks for it to be removed first. The venv is fresh and has none, but the @@ -375,7 +282,6 @@ pip_retry --no-deps --index-url "$FLAGTREE_INDEX_URL" "flagtree===${FLAGTREE_VER # measures the current master, not a pinned snapshot. Override with # TORCH_FL_FLAGGEMS_REVISION to pin a commit for a reproducible run, e.g.: # TORCH_FL_FLAGGEMS_REVISION=$(git ls-remote https://github.com/flagos-ai/FlagGems.git HEAD | cut -f1) -# FLAGGEMS_REVISION="${TORCH_FL_FLAGGEMS_REVISION:-$FLAGGEMS_REVISION_DEFAULT}" FLAGGEMS_REPO="${TORCH_FL_FLAGGEMS_REPO:-$FLAGGEMS_REPO_DEFAULT}" install_flag_gems diff --git a/.github/scripts/set_env_cuda.sh b/.github/scripts/set_env_cuda.sh index 58840839..e8807523 100755 --- a/.github/scripts/set_env_cuda.sh +++ b/.github/scripts/set_env_cuda.sh @@ -28,6 +28,16 @@ REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" # Shared version pins (torch, FlagTree, FlagGems); see .github/version-pins.env. # shellcheck source=.github/version-pins.env source "${REPO_ROOT}/.github/version-pins.env" + +# Shared set_env helpers (pip_retry, FlagGems install, path stripping). +# shellcheck source=.github/scripts/lib/set_env_common.sh +source "${REPO_ROOT}/.github/scripts/lib/set_env_common.sh" + +# CUDA's pip calls target several interpreters (PIP_RETRY_PYTHON at each call) +# and reuse the build cache rather than redownloading the ~2 GB wheel set on +# every retry. +export PIP_RETRY_TIMEOUT=600 +export PIP_RETRY_NO_CACHE=0 CPU_TORCH_VERSION="${TORCH_FL_CPU_TORCH_VERSION:-$CPU_TORCH_VERSION_DEFAULT}" CPU_TORCH_INDEX_URL="${TORCH_FL_CPU_TORCH_INDEX_URL:-$CPU_TORCH_INDEX_URL_DEFAULT}" # FlagTree provides Triton support. The source-free 0.6.2a2 wheel pairs with @@ -56,24 +66,6 @@ FLAGGEMS_CPP_JOBS="${TORCH_FL_FLAGGEMS_CPP_JOBS:-$(nproc 2>/dev/null || echo 4)} VENDOR_MODE="${TORCH_FL_CUDA_VENDOR_MODE:-auto}" VENDOR_TORCH_INDEX_URL="${TORCH_FL_CUDA_VENDOR_TORCH_INDEX_URL:-$VENDOR_TORCH_INDEX_URL_cuda}" -pip_retry() { - local python_exe="$1" - shift - local attempt=1 - while true; do - if "$python_exe" -m pip install --retries 10 --timeout 600 "$@"; then - return 0 - fi - if (( attempt >= 5 )); then - echo "::error::pip install failed after $attempt attempts: $*" - return 1 - fi - echo "::warning::pip install attempt $attempt failed; retrying: $*" - attempt=$((attempt + 1)) - sleep 10 - done -} - find_image_vendor_python() { local candidate="${TORCH_FL_VENDOR_PYTHON:-}" if [[ -n "$candidate" && "$candidate" != */* ]]; then @@ -200,7 +192,7 @@ bootstrap_vendor_python() { "$base_python" -m venv "$vendor_venv" local vendor_python="$vendor_venv/bin/python" "$vendor_python" -m pip install --upgrade pip - pip_retry "$vendor_python" --index-url "$VENDOR_TORCH_INDEX_URL" \ + PIP_RETRY_PYTHON="$vendor_python" pip_retry --index-url "$VENDOR_TORCH_INDEX_URL" \ "torch==$CPU_TORCH_VERSION" VENDOR_PYTHON="$vendor_python" } @@ -446,7 +438,7 @@ if [[ "$(printf '%s\n%s\n' "$FLAGTREE_MIN_GLIBC" "$IMAGE_GLIBC" | sort -V | head fi echo "Image glibc: $IMAGE_GLIBC (FlagTree requires >= $FLAGTREE_MIN_GLIBC)" -pip_retry "$VENV_PYTHON" --no-deps --index-url "$FLAGTREE_INDEX_URL" \ +PIP_RETRY_PYTHON="$VENV_PYTHON" pip_retry --no-deps --index-url "$FLAGTREE_INDEX_URL" \ "flagtree===${FLAGTREE_VERSION}" # Keep only vendor packages that are not provided by FlagTree. In particular, @@ -476,7 +468,7 @@ fi # branch at present; `master` is its default branch and can be overridden with # TORCH_FL_FLAGGEMS_REVISION for reproducible CI experiments. --no-deps keeps # the CPU-only torch ABI intact; install its non-torch dependencies explicitly. -pip_retry "$VENV_PYTHON" packaging 'PyYAML==6.0.1' 'sqlalchemy==2.0.48' numpy +PIP_RETRY_PYTHON="$VENV_PYTHON" pip_retry packaging 'PyYAML==6.0.1' 'sqlalchemy==2.0.48' numpy FLAGGEMS_SOURCE_ROOT="${RUNNER_TEMP:-/tmp}/flag-gems-${CI_STAGE}" rm -rf "$FLAGGEMS_SOURCE_ROOT" git clone --depth 1 --branch master \ @@ -490,7 +482,7 @@ if [[ "$FLAGGEMS_REVISION" != "master" ]]; then fi FLAGGEMS_COMMIT="$(git -C "$FLAGGEMS_SOURCE_ROOT" rev-parse HEAD)" echo "FlagGems source: ${FLAGGEMS_REPOSITORY}@${FLAGGEMS_REVISION} (${FLAGGEMS_COMMIT})" -pip_retry "$VENV_PYTHON" --no-deps --no-build-isolation "$FLAGGEMS_SOURCE_ROOT" +PIP_RETRY_PYTHON="$VENV_PYTHON" pip_retry --no-deps --no-build-isolation "$FLAGGEMS_SOURCE_ROOT" # The FlagGems C++ operators are the native half of FlagGems support: a prebuilt # FlagOS image carries them next to flag_gems, a bootstrapped environment builds @@ -514,13 +506,13 @@ if [[ "$VENDOR_SOURCE" == "bootstrap" ]]; then # parent repository through that provider, and cmake is needed because a CUDA # development image ships a compiler, not a build system. The versions mirror # what the isolated environment installs so both halves of the build agree. - pip_retry "$VENDOR_PYTHON" \ + PIP_RETRY_PYTHON="$VENDOR_PYTHON" pip_retry \ "setuptools>=64,<77" "setuptools-scm>=8,<10" cmake \ "scikit-build-core==0.12.2" "pybind11==3.0.3" "ninja==1.13.0" # Configuring the cpp package probes `import triton` in the building # interpreter and aborts without it, so this environment carries the same # Triton provider the isolated one runs on. - pip_retry "$VENDOR_PYTHON" --no-deps --index-url "$FLAGTREE_INDEX_URL" \ + PIP_RETRY_PYTHON="$VENDOR_PYTHON" pip_retry --no-deps --index-url "$FLAGTREE_INDEX_URL" \ "flagtree===$FLAGTREE_VERSION" # PEP 621 requires a static project name, so the per-vendor suffix is injected # into cpp/pyproject.toml before building (flag-gems-cpp-cuda). @@ -564,7 +556,7 @@ if [[ "$VENDOR_SOURCE" == "bootstrap" ]]; then fi if [[ "$CI_STAGE" == "integration" ]]; then - pip_retry "$VENV_PYTHON" pytest transformers + PIP_RETRY_PYTHON="$VENV_PYTHON" pip_retry pytest transformers fi CPU_TORCH_ROOT="$("$VENV_PYTHON" - <<'PY' @@ -577,26 +569,6 @@ assert torch.version.cuda is None, torch.version.cuda PY )" -strip_vendor_paths() { - local value="${1:-}" - local entry - local -a entries=() - local -a kept=() - IFS=: read -ra entries <<< "$value" - for entry in "${entries[@]}"; do - [[ -z "$entry" ]] && continue - case "$entry" in - "$VENDOR_TORCH_ROOT"|"$VENDOR_TORCH_ROOT"/*) ;; - *) kept+=("$entry") ;; - esac - done - local joined="" - for entry in "${kept[@]}"; do - joined="${joined:+$joined:}$entry" - done - printf '%s' "$joined" -} - export VIRTUAL_ENV="$VENV_ROOT" export PATH="$VENV_ROOT/bin:$PATH" export PYTHONNOUSERSITE=1 diff --git a/.github/scripts/set_env_dcu.sh b/.github/scripts/set_env_dcu.sh index 77968ba9..4b1e10e2 100755 --- a/.github/scripts/set_env_dcu.sh +++ b/.github/scripts/set_env_dcu.sh @@ -28,6 +28,10 @@ REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" # Shared version pins (torch, FlagTree, FlagGems); see .github/version-pins.env. # shellcheck source=.github/version-pins.env source "${REPO_ROOT}/.github/version-pins.env" + +# Shared set_env helpers (pip_retry, FlagGems install, path stripping). +# shellcheck source=.github/scripts/lib/set_env_common.sh +source "${REPO_ROOT}/.github/scripts/lib/set_env_common.sh" CPU_TORCH_INDEX_URL="${TORCH_FL_CPU_TORCH_INDEX_URL:-$CPU_TORCH_INDEX_URL_DEFAULT}" PIP_INDEX_URL_ARG="${TORCH_FL_PIP_INDEX_URL:-$PIP_INDEX_URL_DEFAULT}" @@ -238,102 +242,6 @@ done # # --no-deps on both source packages so pip cannot replace the pinned CPU torch # with something a transitive requirement prefers. -# -# Retries are deliberate: the flagtree wheel is ~350 MB and the shared mirror can -# close a large-wheel response early (IncompleteRead) even though the package is -# there. Retrying just the failed package beats restarting all of setup. -pip_retry() { - local attempt=1 - while true; do - if "$VENV_PYTHON" -m pip install --retries 10 --timeout 300 --no-cache-dir "$@"; then - return 0 - fi - if (( attempt >= 5 )); then - echo "::error::pip install failed after $attempt attempts: $*" - return 1 - fi - echo "::warning::pip install attempt $attempt failed; retrying: $*" - attempt=$((attempt + 1)) - sleep 10 - done -} - -# The FlagGems install is a VCS install, and pip reports the exit status of its -# last step -- the wheel build of whichever tree it managed to fetch. A checkout -# the runner's proxy truncated therefore still ends in `Successfully installed`, -# and pip never notices. On 2026-09-18 the Ascend runner's clone spent ten -# minutes printing -# fatal: unable to access 'https://github.com/flagos-ai/FlagGems.git/': -# Proxy CONNECT aborted -# (364 times), alongside `error: unable to read sha1 file of ...` and -# `error: invalid object 100644 2e574121... for -# '.github/workflows/rule-check.yaml'` for the blobs it never received, and then -# reported -# Successfully installed flaggems_setup-0.0.0 -# -- a 2.1 MB stub named after the build scaffolding rather than the project, -# where the same revision produced the 10 MB -# flag_gems-5.4.0rc2.post1+g437ba3938 on every other platform that ran that -# morning. Nothing failed until the integration suite took its first FlagGems -# route, four minutes later. -# -# So check the install instead of trusting pip's status, and reinstall when it -# is wrong: the conf routes this platform's operators to flagos_python, so an -# unusable flag_gems is not a state this script may leave behind. -flag_gems_installed() { - "$VENV_PYTHON" - "${FLAGGEMS_REVISION:0:9}" <<'PY' -import importlib.metadata as metadata -import importlib.util -import sys - -try: - version = metadata.version("flag_gems") -except metadata.PackageNotFoundError: - raise SystemExit("flag_gems is not installed") - -# A distribution can be installed with no importable package behind it, which -# is what a truncated checkout produces. -if importlib.util.find_spec("flag_gems") is None: - raise SystemExit(f"flag_gems {version} has no importable package") - -print(f"flag_gems {version}") -if sys.argv[1] not in version: - # Not a failure: a revision given as a branch name, or a tarball without - # git metadata, lands on a version string that names neither. Only warn -- - # reinstalling cannot change how the version was written. - print( - f"::warning::flag_gems {version} does not name the requested revision " - f"{sys.argv[1]}; the checkout it was built from may be incomplete", - file=sys.stderr, - ) -PY -} - -# Three attempts at most, and only for an install pip called successful: a pip -# failure has already been retried five times by pip_retry, and repeating that -# spends the job's budget on a link that is down rather than on a bad checkout. -install_flag_gems() { - local attempt=1 - while true; do - if ! pip_retry --no-deps "git+${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}"; then - echo "::error::could not install FlagGems from ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" - return 1 - fi - if flag_gems_installed; then - return 0 - fi - if (( attempt >= 3 )); then - echo "::error::no usable flag_gems after $attempt installs of ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" - return 1 - fi - echo "::warning::install attempt $attempt left no usable flag_gems; reinstalling" - attempt=$((attempt + 1)) - # A truncated tree installs under the build scaffolding's name, so both - # names have to go for the next attempt to be read as a fresh result. - "$VENV_PYTHON" -m pip uninstall -y flag_gems flaggems_setup >/dev/null 2>&1 || true - sleep 10 - done -} - # The venv must not end up with two Tritons. A prebuilt /opt/torch-fl-dcu-venv # baked by an older setup carried the vendor's triton by copy; pip cannot remove # that tree (it ships no dist-info), so it is deleted by path before the flagtree @@ -366,7 +274,6 @@ pip_retry --no-deps --index-url "$FLAGTREE_INDEX_URL" "flagtree===$FLAGTREE_VERS # FlagGems from the flagos-ai fork, tracking master by policy: every CI run # measures the current master, not a pinned snapshot. Override with # TORCH_FL_FLAGGEMS_REVISION to pin a commit for a reproducible run. -# FLAGGEMS_REVISION="${TORCH_FL_FLAGGEMS_REVISION:-$FLAGGEMS_REVISION_DEFAULT}" FLAGGEMS_REPO="${TORCH_FL_FLAGGEMS_REPO:-$FLAGGEMS_REPO_DEFAULT}" install_flag_gems @@ -390,26 +297,6 @@ assert torch.version.cuda is None, torch.version.cuda PY )" -strip_vendor_paths() { - local value="${1:-}" - local entry - local -a entries=() - local -a kept=() - IFS=: read -ra entries <<< "$value" - for entry in "${entries[@]}"; do - [[ -z "$entry" ]] && continue - case "$entry" in - "$VENDOR_TORCH_ROOT"|"$VENDOR_TORCH_ROOT"/*) ;; - *) kept+=("$entry") ;; - esac - done - local joined="" - for entry in "${kept[@]}"; do - joined="${joined:+$joined:}$entry" - done - printf '%s' "$joined" -} - export VIRTUAL_ENV="$VENV_ROOT" export PATH="$VENV_ROOT/bin:$PATH" export PYTHONNOUSERSITE=1 diff --git a/.github/scripts/set_env_gcu.sh b/.github/scripts/set_env_gcu.sh index 74c488a2..0126af13 100755 --- a/.github/scripts/set_env_gcu.sh +++ b/.github/scripts/set_env_gcu.sh @@ -40,6 +40,10 @@ REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" # Shared version pins (torch, FlagTree, FlagGems); see .github/version-pins.env. # shellcheck source=.github/version-pins.env source "${REPO_ROOT}/.github/version-pins.env" + +# Shared set_env helpers (pip_retry, FlagGems install, path stripping). +# shellcheck source=.github/scripts/lib/set_env_common.sh +source "${REPO_ROOT}/.github/scripts/lib/set_env_common.sh" CPU_TORCH_INDEX_URL="${TORCH_FL_CPU_TORCH_INDEX_URL:-$CPU_TORCH_INDEX_URL_DEFAULT}" CPU_TORCH_VERSION="${TORCH_FL_CPU_TORCH_VERSION:-$CPU_TORCH_VERSION_DEFAULT}" PIP_INDEX_URL_ARG="${TORCH_FL_PIP_INDEX_URL:-$PIP_INDEX_URL_DEFAULT}" @@ -149,11 +153,6 @@ else fi VENV_PYTHON="$VENV_ROOT/bin/python" -venv_is_usable() { - [[ -x "$VENV_PYTHON" ]] || return 1 - "$VENV_PYTHON" -m pip --version >/dev/null 2>&1 -} - if ! venv_is_usable; then # The current TopsRider base image does not ship python3.12-venv. Keep the # dependency in the chip-specific setup path so the common workflow remains @@ -211,102 +210,6 @@ PY # # --no-deps on both source packages so pip cannot replace the pinned CPU torch # 2.10 with something a transitive requirement prefers. -# -# Retries are deliberate: the flagtree wheel is large and the shared mirror can -# close the response early (IncompleteRead) even though the package is there. -# Retrying just the failed package beats restarting all of setup. -pip_retry() { - local attempt=1 - while true; do - if "$VENV_PYTHON" -m pip install --retries 10 --timeout 300 --no-cache-dir "$@"; then - return 0 - fi - if (( attempt >= 5 )); then - echo "::error::pip install failed after $attempt attempts: $*" - return 1 - fi - echo "::warning::pip install attempt $attempt failed; retrying: $*" - attempt=$((attempt + 1)) - sleep 10 - done -} - -# The FlagGems install is a VCS install, and pip reports the exit status of its -# last step -- the wheel build of whichever tree it managed to fetch. A checkout -# the runner's proxy truncated therefore still ends in `Successfully installed`, -# and pip never notices. On 2026-09-18 the Ascend runner's clone spent ten -# minutes printing -# fatal: unable to access 'https://github.com/flagos-ai/FlagGems.git/': -# Proxy CONNECT aborted -# (364 times), alongside `error: unable to read sha1 file of ...` and -# `error: invalid object 100644 2e574121... for -# '.github/workflows/rule-check.yaml'` for the blobs it never received, and then -# reported -# Successfully installed flaggems_setup-0.0.0 -# -- a 2.1 MB stub named after the build scaffolding rather than the project, -# where the same revision produced the 10 MB -# flag_gems-5.4.0rc2.post1+g437ba3938 on every other platform that ran that -# morning. Nothing failed until the integration suite took its first FlagGems -# route, four minutes later. -# -# So check the install instead of trusting pip's status, and reinstall when it -# is wrong: the conf routes this platform's operators to flagos_python, so an -# unusable flag_gems is not a state this script may leave behind. -flag_gems_installed() { - "$VENV_PYTHON" - "${FLAGGEMS_REVISION:0:9}" <<'PY' -import importlib.metadata as metadata -import importlib.util -import sys - -try: - version = metadata.version("flag_gems") -except metadata.PackageNotFoundError: - raise SystemExit("flag_gems is not installed") - -# A distribution can be installed with no importable package behind it, which -# is what a truncated checkout produces. -if importlib.util.find_spec("flag_gems") is None: - raise SystemExit(f"flag_gems {version} has no importable package") - -print(f"flag_gems {version}") -if sys.argv[1] not in version: - # Not a failure: a revision given as a branch name, or a tarball without - # git metadata, lands on a version string that names neither. Only warn -- - # reinstalling cannot change how the version was written. - print( - f"::warning::flag_gems {version} does not name the requested revision " - f"{sys.argv[1]}; the checkout it was built from may be incomplete", - file=sys.stderr, - ) -PY -} - -# Three attempts at most, and only for an install pip called successful: a pip -# failure has already been retried five times by pip_retry, and repeating that -# spends the job's budget on a link that is down rather than on a bad checkout. -install_flag_gems() { - local attempt=1 - while true; do - if ! pip_retry --no-deps "git+${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}"; then - echo "::error::could not install FlagGems from ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" - return 1 - fi - if flag_gems_installed; then - return 0 - fi - if (( attempt >= 3 )); then - echo "::error::no usable flag_gems after $attempt installs of ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" - return 1 - fi - echo "::warning::install attempt $attempt left no usable flag_gems; reinstalling" - attempt=$((attempt + 1)) - # A truncated tree installs under the build scaffolding's name, so both - # names have to go for the next attempt to be read as a fresh result. - "$VENV_PYTHON" -m pip uninstall -y flag_gems flaggems_setup >/dev/null 2>&1 || true - sleep 10 - done -} - # An NVIDIA triton wheel must not be present: it carries no "enflame" backend, # so flag_gems' backend discovery fails and every FlagGems route raises at # import. flagtree installs its own `triton` package under the same import name @@ -336,7 +239,6 @@ pip_retry --no-deps --index-url "$FLAGTREE_INDEX_URL" "flagtree===$FLAGTREE_VERS # TORCH_FL_FLAGGEMS_REVISION to pin a commit for a reproducible run. The routing # measured against the old 3c6f7537d pin is recorded in # docs/vendors/gcu/flaggems-test-results.md. -# FLAGGEMS_REVISION="${TORCH_FL_FLAGGEMS_REVISION:-$FLAGGEMS_REVISION_DEFAULT}" FLAGGEMS_REPO="${TORCH_FL_FLAGGEMS_REPO:-$FLAGGEMS_REPO_DEFAULT}" install_flag_gems diff --git a/.github/scripts/set_env_metax.sh b/.github/scripts/set_env_metax.sh index af8ec420..dfedc9be 100755 --- a/.github/scripts/set_env_metax.sh +++ b/.github/scripts/set_env_metax.sh @@ -29,100 +29,9 @@ REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" # shellcheck source=.github/version-pins.env source "${REPO_ROOT}/.github/version-pins.env" -# Retries are deliberate: the flagtree wheel is ~180MB and the shared mirror can -# close a large-wheel response early (IncompleteRead) even though the package is -# there. Retrying just the failed package beats restarting all of setup. -pip_retry() { - local attempt=1 - while true; do - if python -m pip install --retries 10 --timeout 300 --no-cache-dir "$@"; then - return 0 - fi - if (( attempt >= 5 )); then - echo "::error::pip install failed after $attempt attempts: $*" - return 1 - fi - echo "::warning::pip install attempt $attempt failed; retrying: $*" - attempt=$((attempt + 1)) - sleep 10 - done -} - -# The FlagGems install is a VCS install, and pip reports the exit status of its -# last step -- the wheel build of whichever tree it managed to fetch. A checkout -# the runner's proxy truncated therefore still ends in `Successfully installed`, -# and pip never notices. On 2026-09-18 the Ascend runner's clone spent ten -# minutes printing -# fatal: unable to access 'https://github.com/flagos-ai/FlagGems.git/': -# Proxy CONNECT aborted -# (364 times), alongside `error: unable to read sha1 file of ...` and -# `error: invalid object 100644 2e574121... for -# '.github/workflows/rule-check.yaml'` for the blobs it never received, and then -# reported -# Successfully installed flaggems_setup-0.0.0 -# -- a 2.1 MB stub named after the build scaffolding rather than the project, -# where the same revision produced the 10 MB -# flag_gems-5.4.0rc2.post1+g437ba3938 on every other platform that ran that -# morning. Nothing failed until the integration suite took its first FlagGems -# route, four minutes later. -# -# So check the install instead of trusting pip's status, and reinstall when it -# is wrong: the conf routes this platform's operators to flagos_python, so an -# unusable flag_gems is not a state this script may leave behind. -flag_gems_installed() { - python - "${FLAGGEMS_REVISION:0:9}" <<'PY' -import importlib.metadata as metadata -import importlib.util -import sys - -try: - version = metadata.version("flag_gems") -except metadata.PackageNotFoundError: - raise SystemExit("flag_gems is not installed") - -# A distribution can be installed with no importable package behind it, which -# is what a truncated checkout produces. -if importlib.util.find_spec("flag_gems") is None: - raise SystemExit(f"flag_gems {version} has no importable package") - -print(f"flag_gems {version}") -if sys.argv[1] not in version: - # Not a failure: a revision given as a branch name, or a tarball without - # git metadata, lands on a version string that names neither. Only warn -- - # reinstalling cannot change how the version was written. - print( - f"::warning::flag_gems {version} does not name the requested revision " - f"{sys.argv[1]}; the checkout it was built from may be incomplete", - file=sys.stderr, - ) -PY -} - -# Three attempts at most, and only for an install pip called successful: a pip -# failure has already been retried five times by pip_retry, and repeating that -# spends the job's budget on a link that is down rather than on a bad checkout. -install_flag_gems() { - local attempt=1 - while true; do - if ! pip_retry --no-deps "git+${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}"; then - echo "::error::could not install FlagGems from ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" - return 1 - fi - if flag_gems_installed; then - return 0 - fi - if (( attempt >= 3 )); then - echo "::error::no usable flag_gems after $attempt installs of ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" - return 1 - fi - echo "::warning::install attempt $attempt left no usable flag_gems; reinstalling" - attempt=$((attempt + 1)) - # A truncated tree installs under the build scaffolding's name, so both - # names have to go for the next attempt to be read as a fresh result. - python -m pip uninstall -y flag_gems flaggems_setup >/dev/null 2>&1 || true - sleep 10 - done -} +# Shared set_env helpers (pip_retry, FlagGems install, path stripping). +# shellcheck source=.github/scripts/lib/set_env_common.sh +source "${REPO_ROOT}/.github/scripts/lib/set_env_common.sh" # Resolve the operator names the checked-in Python kernels call against the # FlagGems an interpreter in $1 imports, and print "/". diff --git a/.github/scripts/set_env_musa.sh b/.github/scripts/set_env_musa.sh index 8bf7ce0f..09ec0e1c 100755 --- a/.github/scripts/set_env_musa.sh +++ b/.github/scripts/set_env_musa.sh @@ -40,6 +40,10 @@ REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" # Shared version pins (torch, FlagTree, FlagGems); see .github/version-pins.env. # shellcheck source=.github/version-pins.env source "${REPO_ROOT}/.github/version-pins.env" + +# Shared set_env helpers (pip_retry, FlagGems install, path stripping). +# shellcheck source=.github/scripts/lib/set_env_common.sh +source "${REPO_ROOT}/.github/scripts/lib/set_env_common.sh" CPU_TORCH_INDEX_URL="${TORCH_FL_CPU_TORCH_INDEX_URL:-$CPU_TORCH_INDEX_URL_DEFAULT}" CPU_TORCH_VERSION="${TORCH_FL_CPU_TORCH_VERSION:-$CPU_TORCH_VERSION_DEFAULT}" PIP_INDEX_URL_ARG="${TORCH_FL_PIP_INDEX_URL:-$PIP_INDEX_URL_DEFAULT}" @@ -141,11 +145,6 @@ else fi VENV_PYTHON="$VENV_ROOT/bin/python" -venv_is_usable() { - [[ -x "$VENV_PYTHON" ]] || return 1 - "$VENV_PYTHON" -m pip --version >/dev/null 2>&1 -} - if ! venv_is_usable; then # The vendor base image may not ship the matching python*-venv package. Keep # that dependency in the chip-specific setup path so the common workflow stays @@ -255,101 +254,6 @@ fi # --no-deps on both source packages so pip cannot replace the pinned CPU torch # 2.10 with something a transitive requirement prefers. # -# Retries are deliberate: the flagtree wheel is 180 MB and the shared mirror can -# close a large-wheel response early (IncompleteRead) even though the package is -# there. Retrying just the failed package beats restarting all of setup. -pip_retry() { - local attempt=1 - while true; do - if "$VENV_PYTHON" -m pip install --retries 10 --timeout 300 --no-cache-dir "$@"; then - return 0 - fi - if (( attempt >= 5 )); then - echo "::error::pip install failed after $attempt attempts: $*" - return 1 - fi - echo "::warning::pip install attempt $attempt failed; retrying: $*" - attempt=$((attempt + 1)) - sleep 10 - done -} - -# The FlagGems install is a VCS install, and pip reports the exit status of its -# last step -- the wheel build of whichever tree it managed to fetch. A checkout -# the runner's proxy truncated therefore still ends in `Successfully installed`, -# and pip never notices. On 2026-09-18 the Ascend runner's clone spent ten -# minutes printing -# fatal: unable to access 'https://github.com/flagos-ai/FlagGems.git/': -# Proxy CONNECT aborted -# (364 times), alongside `error: unable to read sha1 file of ...` and -# `error: invalid object 100644 2e574121... for -# '.github/workflows/rule-check.yaml'` for the blobs it never received, and then -# reported -# Successfully installed flaggems_setup-0.0.0 -# -- a 2.1 MB stub named after the build scaffolding rather than the project, -# where the same revision produced the 10 MB -# flag_gems-5.4.0rc2.post1+g437ba3938 on every other platform that ran that -# morning. Nothing failed until the integration suite took its first FlagGems -# route, four minutes later. -# -# So check the install instead of trusting pip's status, and reinstall when it -# is wrong: the conf routes this platform's operators to flagos_python, so an -# unusable flag_gems is not a state this script may leave behind. -flag_gems_installed() { - "$VENV_PYTHON" - "${FLAGGEMS_REVISION:0:9}" <<'PY' -import importlib.metadata as metadata -import importlib.util -import sys - -try: - version = metadata.version("flag_gems") -except metadata.PackageNotFoundError: - raise SystemExit("flag_gems is not installed") - -# A distribution can be installed with no importable package behind it, which -# is what a truncated checkout produces. -if importlib.util.find_spec("flag_gems") is None: - raise SystemExit(f"flag_gems {version} has no importable package") - -print(f"flag_gems {version}") -if sys.argv[1] not in version: - # Not a failure: a revision given as a branch name, or a tarball without - # git metadata, lands on a version string that names neither. Only warn -- - # reinstalling cannot change how the version was written. - print( - f"::warning::flag_gems {version} does not name the requested revision " - f"{sys.argv[1]}; the checkout it was built from may be incomplete", - file=sys.stderr, - ) -PY -} - -# Three attempts at most, and only for an install pip called successful: a pip -# failure has already been retried five times by pip_retry, and repeating that -# spends the job's budget on a link that is down rather than on a bad checkout. -install_flag_gems() { - local attempt=1 - while true; do - if ! pip_retry --no-deps "git+${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}"; then - echo "::error::could not install FlagGems from ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" - return 1 - fi - if flag_gems_installed; then - return 0 - fi - if (( attempt >= 3 )); then - echo "::error::no usable flag_gems after $attempt installs of ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" - return 1 - fi - echo "::warning::install attempt $attempt left no usable flag_gems; reinstalling" - attempt=$((attempt + 1)) - # A truncated tree installs under the build scaffolding's name, so both - # names have to go for the next attempt to be read as a fresh result. - "$VENV_PYTHON" -m pip uninstall -y flag_gems flaggems_setup >/dev/null 2>&1 || true - sleep 10 - done -} - # flagtree is the Triton build carrying the "mthreads" backend. 3.6 is not a # preference but a requirement: current FlagGems uses tl.map_elementwise and # triton.knobs, which flagtree 0.5.x (Triton 3.1) does not have -- that pair @@ -365,7 +269,6 @@ pip_retry --no-deps --index-url "$FLAGTREE_INDEX_URL" "flagtree===$FLAGTREE_VERS # FlagGems from the flagos-ai fork, tracking master by policy: every CI run # measures the current master, not a pinned snapshot. Override with # TORCH_FL_FLAGGEMS_REVISION to pin a commit for a reproducible run. -# FLAGGEMS_REVISION="${TORCH_FL_FLAGGEMS_REVISION:-$FLAGGEMS_REVISION_DEFAULT}" FLAGGEMS_REPO="${TORCH_FL_FLAGGEMS_REPO:-$FLAGGEMS_REPO_DEFAULT}" install_flag_gems diff --git a/.github/scripts/set_env_ppu.sh b/.github/scripts/set_env_ppu.sh index dc82e5cb..8a76a1a5 100644 --- a/.github/scripts/set_env_ppu.sh +++ b/.github/scripts/set_env_ppu.sh @@ -50,6 +50,10 @@ REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)" # Shared version pins (torch, FlagTree, FlagGems); see .github/version-pins.env. # shellcheck source=.github/version-pins.env source "${REPO_ROOT}/.github/version-pins.env" + +# Shared set_env helpers (pip_retry, FlagGems install, path stripping). +# shellcheck source=.github/scripts/lib/set_env_common.sh +source "${REPO_ROOT}/.github/scripts/lib/set_env_common.sh" CPU_TORCH_VERSION="${TORCH_FL_CPU_TORCH_VERSION:-$CPU_TORCH_VERSION_DEFAULT}" # Use the official indexes by default: the PPU runner pod's HTTP proxy returns # 500 on HTTPS CONNECT to *.tuna.tsinghua.edu.cn, so the Tsinghua PyPI and @@ -77,7 +81,6 @@ FLAGTREE_INDEX_URL="${TORCH_FL_FLAGTREE_INDEX_URL:-$FLAGTREE_INDEX_URL_DEFAULT}" # that mount was not present on every runner pod, and its absence aborted the # job in environment setup (see the install below). `master` by request; # override with TORCH_FL_FLAGGEMS_REVISION to pin a commit for a reproducible run. -# FLAGGEMS_REVISION="${TORCH_FL_FLAGGEMS_REVISION:-$FLAGGEMS_REVISION_DEFAULT}" FLAGGEMS_REPO="${TORCH_FL_FLAGGEMS_REPO:-$FLAGGEMS_REPO_DEFAULT}" @@ -245,102 +248,6 @@ fi # index-url choice and applies regardless of which index ends up serving it. export PIP_DEFAULT_TIMEOUT=120 -# FlagTree is a ~363 MB wheel served by a shared mirror, and the FlagGems git -# install has to resolve github.com through the same proxy. Both are large -# enough that a single transient reset would otherwise fail the job in setup, -# so retry them the way set_env_musa.sh does. -pip_retry() { - local attempt=1 - while true; do - if "$VENV_PYTHON" -m pip install --retries 10 --timeout 300 --no-cache-dir "$@"; then - return 0 - fi - if ((attempt >= 5)); then - echo "::error::pip install failed after $attempt attempts: $*" - return 1 - fi - echo "::warning::pip install attempt $attempt failed; retrying: $*" - attempt=$((attempt + 1)) - sleep 10 - done -} - -# The FlagGems install is a VCS install, and pip reports the exit status of its -# last step -- the wheel build of whichever tree it managed to fetch. A checkout -# the runner's proxy truncated therefore still ends in `Successfully installed`, -# and pip never notices. On 2026-09-18 the Ascend runner's clone spent ten -# minutes printing -# fatal: unable to access 'https://github.com/flagos-ai/FlagGems.git/': -# Proxy CONNECT aborted -# (364 times), alongside `error: unable to read sha1 file of ...` and -# `error: invalid object 100644 2e574121... for -# '.github/workflows/rule-check.yaml'` for the blobs it never received, and then -# reported -# Successfully installed flaggems_setup-0.0.0 -# -- a 2.1 MB stub named after the build scaffolding rather than the project, -# where the same revision produced the 10 MB -# flag_gems-5.4.0rc2.post1+g437ba3938 on every other platform that ran that -# morning. Nothing failed until the integration suite took its first FlagGems -# route, four minutes later. -# -# So check the install instead of trusting pip's status, and reinstall when it -# is wrong: the conf routes this platform's operators to flagos_python, so an -# unusable flag_gems is not a state this script may leave behind. -flag_gems_installed() { - "$VENV_PYTHON" - "${FLAGGEMS_REVISION:0:9}" <<'PY' -import importlib.metadata as metadata -import importlib.util -import sys - -try: - version = metadata.version("flag_gems") -except metadata.PackageNotFoundError: - raise SystemExit("flag_gems is not installed") - -# A distribution can be installed with no importable package behind it, which -# is what a truncated checkout produces. -if importlib.util.find_spec("flag_gems") is None: - raise SystemExit(f"flag_gems {version} has no importable package") - -print(f"flag_gems {version}") -if sys.argv[1] not in version: - # Not a failure: a revision given as a branch name, or a tarball without - # git metadata, lands on a version string that names neither. Only warn -- - # reinstalling cannot change how the version was written. - print( - f"::warning::flag_gems {version} does not name the requested revision " - f"{sys.argv[1]}; the checkout it was built from may be incomplete", - file=sys.stderr, - ) -PY -} - -# Three attempts at most, and only for an install pip called successful: a pip -# failure has already been retried five times by pip_retry, and repeating that -# spends the job's budget on a link that is down rather than on a bad checkout. -install_flag_gems() { - local attempt=1 - while true; do - if ! pip_retry --no-deps "git+${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}"; then - echo "::error::could not install FlagGems from ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" - return 1 - fi - if flag_gems_installed; then - return 0 - fi - if (( attempt >= 3 )); then - echo "::error::no usable flag_gems after $attempt installs of ${FLAGGEMS_REPO}@${FLAGGEMS_REVISION}" - return 1 - fi - echo "::warning::install attempt $attempt left no usable flag_gems; reinstalling" - attempt=$((attempt + 1)) - # A truncated tree installs under the build scaffolding's name, so both - # names have to go for the next attempt to be read as a fresh result. - "$VENV_PYTHON" -m pip uninstall -y flag_gems flaggems_setup >/dev/null 2>&1 || true - sleep 10 - done -} - # --- Reaching a host around the runner's HTTP proxy -------------------------- # The runner injects HTTP(S)_PROXY into the job container along with its own # allowlist and NO_PROXY list (this pod: localhost,127.0.0.1,10.1.12.192, @@ -551,26 +458,6 @@ assert torch.version.cuda is None, torch.version.cuda PY )" -strip_vendor_paths() { - local value="${1:-}" - local entry - local -a entries=() - local -a kept=() - IFS=: read -ra entries <<< "$value" - for entry in "${entries[@]}"; do - [[ -z "$entry" ]] && continue - case "$entry" in - "$VENDOR_TORCH_ROOT"|"$VENDOR_TORCH_ROOT"/*) ;; - *) kept+=("$entry") ;; - esac - done - local joined="" - for entry in "${kept[@]}"; do - joined="${joined:+$joined:}$entry" - done - printf '%s' "$joined" -} - export VIRTUAL_ENV="$VENV_ROOT" export PATH="$VENV_ROOT/bin:$PATH" export PYTHONNOUSERSITE=1 diff --git a/.github/workflows/agnostic-checks.yml b/.github/workflows/agnostic-checks.yml index c4d45a27..521c751d 100644 --- a/.github/workflows/agnostic-checks.yml +++ b/.github/workflows/agnostic-checks.yml @@ -70,7 +70,7 @@ jobs: - name: Check setup-script syntax run: | set -euo pipefail - for script in .github/scripts/set_env_*.sh .github/version-pins.env; do + for script in .github/scripts/set_env_*.sh .github/scripts/lib/set_env_common.sh .github/version-pins.env; do bash -n "$script" echo "OK $script" done @@ -93,6 +93,7 @@ jobs: tests/unit/test_platform_support_detect.py \ tests/unit/test_run_integration_tests.py \ tests/unit/test_run_unit_tests.py \ + tests/unit/test_set_env_common.py \ tests/unit/test_transformers_automation.py \ tests/unit/test_transformers_hf_tests.py \ tests/unit/test_transformers_model_probe.py \ diff --git a/tests/unit/test_set_env_common.py b/tests/unit/test_set_env_common.py new file mode 100644 index 00000000..f2207d8a --- /dev/null +++ b/tests/unit/test_set_env_common.py @@ -0,0 +1,79 @@ +# Copyright 2026 FlagOS Contributors +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""The set_env_*.sh scripts share one helper library. + +`pip_retry`, `flag_gems_installed`, `install_flag_gems`, `strip_vendor_paths` +and `venv_is_usable` used to be copied into every platform script -- `pip_retry` +in five slightly different spellings -- so a fix had to be applied seven times +and was easy to miss. They now live in `.github/scripts/lib/set_env_common.sh`, +which every script sources. This keeps it that way: the library defines them, +no script redefines them, and every script sources the library. + +Pure text: the scripts cannot be executed here (they provision vendor +environments), but the wiring is fully visible in their source. +""" + +import re +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[2] +SCRIPTS_DIR = REPO_ROOT / ".github" / "scripts" +LIB = SCRIPTS_DIR / "lib" / "set_env_common.sh" + +SHARED_FUNCS = [ + "pip_retry", + "flag_gems_installed", + "install_flag_gems", + "strip_vendor_paths", + "venv_is_usable", +] + +SOURCE_LINE = 'source "${REPO_ROOT}/.github/scripts/lib/set_env_common.sh"' + + +def _scripts() -> list[Path]: + return sorted(SCRIPTS_DIR.glob("set_env_*.sh")) + + +def test_the_library_defines_every_shared_function(): + text = LIB.read_text(encoding="utf-8") + for fn in SHARED_FUNCS: + assert re.search(rf"^{fn}\(\) \{{", text, re.M), fn + + +def test_no_script_redefines_a_shared_function(): + for script in _scripts(): + text = script.read_text(encoding="utf-8") + for fn in SHARED_FUNCS: + assert not re.search(rf"^{fn}\(\) \{{", text, re.M), f"{script.name}: {fn}" + + +def test_every_script_sources_the_library(): + for script in _scripts(): + text = script.read_text(encoding="utf-8") + assert SOURCE_LINE in text, script.name + + +def test_the_cuda_script_still_selects_its_interpreter_per_call(): + """CUDA drives several interpreters, so it must set PIP_RETRY_PYTHON. + + The shared pip_retry defaults to $VENV_PYTHON; CUDA's bootstrap/vendor steps + need a different one, so every call has to name it or the wrong interpreter + is used silently. + """ + text = (SCRIPTS_DIR / "set_env_cuda.sh").read_text(encoding="utf-8") + calls = re.findall(r"pip_retry ", text) + prefixed = re.findall(r'PIP_RETRY_PYTHON="\$[A-Za-z_]+" pip_retry ', text) + assert len(calls) == len(prefixed) == 7, (len(calls), len(prefixed))