diff --git a/.github/workflows/CICD.yml b/.github/workflows/CICD.yml index 3d134a5a21e..3181ff53f4b 100644 --- a/.github/workflows/CICD.yml +++ b/.github/workflows/CICD.yml @@ -2,7 +2,7 @@ name: CICD # spell-checker:ignore (abbrev/names) CACHEDIR CICD CodeCOV MacOS MinGW MSVC musl taiki # spell-checker:ignore (env/flags) Awarnings Ccodegen Coverflow Cpanic Dwarnings RUSTDOCFLAGS RUSTFLAGS Zpanic CARGOFLAGS CLEVEL nodocs -# spell-checker:ignore (jargon) SHAs deps dequote softprops subshell toolchain fuzzers dedupe devel profdata profraw +# spell-checker:ignore (jargon) SHAs deps dequote softprops subshell toolchain fuzzers dedupe devel profdata profraw Cprofile # spell-checker:ignore (people) Peltoche rivy Anson dawidd # spell-checker:ignore (shell/tools) binutils choco clippy dmake esac fakeroot fdesc fdescfs gmake grcov halium lcov libclang libcrypto libfuse libssl limactl nextest nocross pacman popd printf pushd redoxer rsync rustc rustfmt rustup shopt sccache setfacl utmpdump xargs zstd # spell-checker:ignore (misc) aarch alnum armhf bindir busytest coreutils defconfig DESTDIR gecos getenforce gnueabihf issuecomment maint manpages msys multisize noconfirm nofeatures nullglob onexitbegin onexitend pell runtest tempfile testsuite toybox uutils libsystemd codspeed wasip libexecinfo oniguruma sysroot @@ -635,6 +635,40 @@ jobs: ${{ steps.dep_vars.outputs.CARGO_UTILITY_LIST_OPTIONS }} -p coreutils env: RUST_BACKTRACE: "1" + - name: Install llvm-tools (PGO) + if: | + matrix.job.skip-publish != true && matrix.job.check-only != true && + matrix.job.use-cross != 'use-cross' && + (matrix.job.target == 'x86_64-unknown-linux-gnu' || matrix.job.target == 'aarch64-unknown-linux-gnu') + shell: bash + run: rustup component add llvm-tools + - name: Train PGO profiles + if: | + matrix.job.skip-publish != true && matrix.job.check-only != true && + matrix.job.use-cross != 'use-cross' && + (matrix.job.target == 'x86_64-unknown-linux-gnu' || matrix.job.target == 'aarch64-unknown-linux-gnu') + shell: bash + run: | + ./util/build-pgo.sh \ + --target-dir "${{ github.workspace }}/target/coreutils-pgo" \ + --features "${{ matrix.job.features }}" \ + --train-only + echo "RUSTFLAGS=${RUSTFLAGS:+${RUSTFLAGS} }-Cprofile-use=${{ github.workspace }}/target/coreutils-pgo/coreutils.profdata" >> "$GITHUB_ENV" + - name: Verify PGO is applied to the published build + if: | + matrix.job.skip-publish != true && matrix.job.check-only != true && + matrix.job.use-cross != 'use-cross' && + (matrix.job.target == 'x86_64-unknown-linux-gnu' || matrix.job.target == 'aarch64-unknown-linux-gnu') + shell: bash + run: | + ## The release artifact must be the PGO build: if RUSTFLAGS did not + ## survive GITHUB_ENV we would silently publish an unoptimized binary. + echo "RUSTFLAGS=${RUSTFLAGS}" + case "${RUSTFLAGS}" in + *-Cprofile-use=*) ;; + *) echo "::error::-Cprofile-use missing from RUSTFLAGS" ; exit 1 ;; + esac + test -s "${{ github.workspace }}/target/coreutils-pgo/coreutils.profdata" - name: Build coreutils shell: bash if: matrix.job.skip-publish != true && matrix.job.check-only != true && matrix.job.target != 'x86_64-pc-windows-msvc' diff --git a/util/build-pgo.sh b/util/build-pgo.sh new file mode 100755 index 00000000000..b7a6292180d --- /dev/null +++ b/util/build-pgo.sh @@ -0,0 +1,236 @@ +#!/usr/bin/env bash +# spell-checker:ignore (jargon) profdata profraw sysroot rustlib nullglob aeiou nocheck CGU mktemp Cprofile awk +# +# Build uutils coreutils with Profile-Guided Optimization. +# +# 1. build an instrumented multicall binary (-Cprofile-generate) +# 2. run representative workloads to collect raw profiles +# 3. merge them with llvm-profdata +# 4. build the optimized binary (-Cprofile-use), unless --train-only +# +# Usage: +# util/build-pgo.sh [--target-dir DIR] [--features LIST] [--train-only] +# [--llvm-profdata PATH] + +set -euo pipefail + +REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +TARGET_DIR="${REPO_ROOT}/target/coreutils-pgo" +FEATURES="unix" +TRAIN_ONLY=0 +LLVM_PROFDATA="" + +while [ $# -gt 0 ]; do + case "$1" in + --target-dir) TARGET_DIR="$2"; shift 2 ;; + --features) FEATURES="$2"; shift 2 ;; + --llvm-profdata) LLVM_PROFDATA="$2"; shift 2 ;; + --train-only) TRAIN_ONLY=1; shift ;; + -h|--help) sed -n '3,14p' "${BASH_SOURCE[0]}"; exit 0 ;; + *) echo "unknown argument: $1" >&2; exit 2 ;; + esac +done + +mkdir -p "$TARGET_DIR" +TARGET_DIR="$(cd "$TARGET_DIR" && pwd)" + +SCRIPT_START=$SECONDS + +fmt_duration() { printf '%dm%02ds' $(($1 / 60)) $(($1 % 60)); } +begin_step() { + STEP_NAME="$1" + STEP_START=$SECONDS + echo + echo "=== ${STEP_NAME} ===" +} +end_step() { echo "--- ${STEP_NAME}: $(fmt_duration $((SECONDS - STEP_START))) ---"; } + +PROFILE_DIR="${TARGET_DIR}/profiles" +CORPUS_DIR="${TARGET_DIR}/corpus" +MERGED="${TARGET_DIR}/coreutils.profdata" + +# llvm-profdata must come from the *active* toolchain: its version has to match +# the rustc that instrumented the binary. +if [ -z "$LLVM_PROFDATA" ]; then + HOST="$(rustc --print host-tuple)" + LLVM_PROFDATA="$(rustc --print sysroot)/lib/rustlib/${HOST}/bin/llvm-profdata" +fi +if [ ! -x "$LLVM_PROFDATA" ]; then + echo "llvm-profdata not found at ${LLVM_PROFDATA}" >&2 + echo "Run: rustup component add llvm-tools" >&2 + exit 1 +fi +echo "llvm-profdata: ${LLVM_PROFDATA}" + +cargo_build() { + # $1: target dir, $2: extra rustflags + local feature_args=() + [ -n "$FEATURES" ] && feature_args=(--features="$FEATURES") + ( + cd "$REPO_ROOT" + export CARGO_TARGET_DIR="$1" + export CARGO_INCREMENTAL=0 + export RUSTFLAGS="${RUSTFLAGS:+${RUSTFLAGS} }$2" + echo "Running: cargo build --release ${feature_args[*]}" + echo " RUSTFLAGS=${RUSTFLAGS}" + cargo build --release "${feature_args[@]}" + ) +} + +begin_step "Step 1: instrumented build" +INSTR_DIR="${TARGET_DIR}/instrumented" +rm -rf "$PROFILE_DIR" +mkdir -p "$PROFILE_DIR" +# The instrumented binary is only ever run for training, so skip the expensive +# whole-program codegen. Profiles are keyed by function, not by LTO/CGU layout, +# so this does not affect the profile the final build consumes. +( + export CARGO_PROFILE_RELEASE_LTO=false + export CARGO_PROFILE_RELEASE_CODEGEN_UNITS=16 + cargo_build "$INSTR_DIR" "-Cprofile-generate=${PROFILE_DIR}" +) + +BIN="${INSTR_DIR}/release/coreutils" +[ -x "$BIN" ] || { echo "instrumented binary not found: ${BIN}" >&2; exit 1; } +end_step + +begin_step "Step 2: corpus" +# Everything is generated from scratch: reading whatever /usr/share/dict/words +# or /etc/passwd happens to hold would make the profile machine-dependent. +export LLVM_PROFILE_FILE="${PROFILE_DIR}/coreutils-%p.profraw" +rm -rf "$CORPUS_DIR" +mkdir -p "$CORPUS_DIR" + +WORDS="${CORPUS_DIR}/words.txt" +NUMBERS="${CORPUS_DIR}/numbers.txt" +REPEATED="${CORPUS_DIR}/repeated.txt" +PAIRS="${CORPUS_DIR}/pairs.txt" +BLOB="${CORPUS_DIR}/blob.bin" +COLUMNS="${CORPUS_DIR}/columns.txt" + +# Shuffled by a stride so sort/uniq get unordered input rather than a no-op. +awk 'BEGIN { for (i = 0; i < 200000; i++) printf "w%06d\n", (i * 7919) % 200000 }' > "$WORDS" +awk 'BEGIN { for (i = 0; i < 100000; i++) printf "line%04d\n", i % 1000 }' > "$REPEATED" +awk 'BEGIN { for (i = 0; i < 100000; i++) printf "%08d value%d\n", i, i }' > "$PAIRS" +awk 'BEGIN { for (i = 0; i < 2000; i++) printf "user%d:x:%d:%d:User %d:/home/user%d:/bin/sh\n", i, 1000+i, 1000+i, i, i }' > "$COLUMNS" +# seq/dd are part of what we want to profile, so use the instrumented binary. +"$BIN" seq 500000 > "$NUMBERS" +"$BIN" dd "if=${BIN}" "of=${BLOB}" bs=64K count=64 2>/dev/null +end_step + +begin_step "Step 3: training workloads" +WORK="$(mktemp -d)" +trap 'rm -rf "$WORK"' EXIT + +# Individual workloads are allowed to fail (a util may be absent from the +# selected feature set); a partial profile is still a usable profile. +run() { "$BIN" "$@" >/dev/null 2>&1 || true; } + +# sort: the single hottest util, in its main modes +run sort "$WORDS" -o "${WORK}/sorted.txt" +run sort -n "$NUMBERS" -o "${WORK}/sorted-n.txt" +run sort -u "$REPEATED" -o "${WORK}/sorted-u.txt" +run sort -r "$WORDS" -o "${WORK}/sorted-r.txt" +run sort -k2,2 -t: "$COLUMNS" -o "${WORK}/sorted-k.txt" + +# wc / cat / head / tail, to a regular file so the write path is not /dev/null +run wc -l "$WORDS" +run wc -w "$WORDS" +run wc -c "$BLOB" +run wc "$WORDS" "$NUMBERS" +"$BIN" cat "$WORDS" > "${WORK}/cat.txt" 2>/dev/null || true +"$BIN" cat -n "$WORDS" > "${WORK}/cat-n.txt" 2>/dev/null || true +"$BIN" head -n 50000 "$WORDS" > "${WORK}/head.txt" 2>/dev/null || true +"$BIN" tail -n 50000 "$WORDS" > "${WORK}/tail.txt" 2>/dev/null || true +run head -c 1M "$BLOB" + +# Pipelines: splice/copy_file_range fast paths only fire on a pipe, so they are +# never reached when every workload redirects to /dev/null. +"$BIN" cat "$WORDS" | "$BIN" sort | "$BIN" uniq -c > "${WORK}/pipe1.txt" 2>/dev/null || true +"$BIN" seq 200000 | "$BIN" wc -l > /dev/null 2>&1 || true +"$BIN" cat "$BLOB" | "$BIN" sha256sum > /dev/null 2>&1 || true +"$BIN" sort -n "$NUMBERS" | "$BIN" tail -n 1000 | "$BIN" cut -c1-4 > "${WORK}/pipe2.txt" 2>/dev/null || true + +# text utils +run uniq "$REPEATED" +run uniq -c "$REPEATED" +run uniq -u "$REPEATED" +run cut -d: -f1,3 "$COLUMNS" +run cut -c1-10 "$WORDS" +run nl "$WORDS" +run fold -w 40 "$WORDS" +run expand "$WORDS" +run unexpand "$WORDS" +run paste "$WORDS" "$WORDS" +run join "$PAIRS" "$PAIRS" +run join --nocheck-order "$PAIRS" "$PAIRS" +run split -l 20000 "$WORDS" "${WORK}/split-" +"$BIN" tr a-z A-Z < "$WORDS" > "${WORK}/tr.txt" 2>/dev/null || true +"$BIN" tr -d aeiou < "$WORDS" > "${WORK}/tr-d.txt" 2>/dev/null || true +"$BIN" tee "${WORK}/tee.txt" < "$NUMBERS" > /dev/null 2>&1 || true + +# numbers / encoding / hashing +run seq 1000000 +run seq 0 0.1 10000 +"$BIN" base64 "$BLOB" > "${WORK}/encoded.b64" 2>/dev/null || true +run base64 -d "${WORK}/encoded.b64" +run cksum "$BLOB" +run md5sum "$BLOB" +run sha1sum "$BLOB" +run sha256sum "$BLOB" + +# file management: cp/mv/rm/ls are as hot in practice as anything above +run cp "$WORDS" "${WORK}/copy.txt" +run cp -r "${REPO_ROOT}/src/uu/sort" "${WORK}/tree" +run mv "${WORK}/copy.txt" "${WORK}/moved.txt" +run mv "${WORK}/tree" "${WORK}/tree2" +run ls -la "${REPO_ROOT}/src/uu" +run ls -lR "${REPO_ROOT}/src/uu" +run ls --color=always -la "${REPO_ROOT}/src" +run du -sh "${REPO_ROOT}/src" +run df -h +run rm -rf "${WORK}/tree2" + +echo "Workloads complete." +end_step + +begin_step "Step 4: merging profiles" +shopt -s nullglob +RAW=("${PROFILE_DIR}"/coreutils-*.profraw) +shopt -u nullglob +if [ ${#RAW[@]} -eq 0 ]; then + echo "no .profraw files found in ${PROFILE_DIR}" >&2 + exit 1 +fi +echo "Merging ${#RAW[@]} profile(s)..." +"$LLVM_PROFDATA" merge -sparse "${RAW[@]}" -o "$MERGED" +echo "Merged profile: ${MERGED}" + +# A profile that covers almost nothing still builds fine and silently produces a +# barely-optimized binary, so fail loudly instead: every workload in step 3 is +# allowed to fail individually, and without this a broken corpus would ship. +"$LLVM_PROFDATA" show "$MERGED" | head -6 +COVERED="$("$LLVM_PROFDATA" show "$MERGED" | awk '/^Total functions:/ { print $3 }')" +MIN_FUNCTIONS=500 +if [ -z "$COVERED" ] || [ "$COVERED" -lt "$MIN_FUNCTIONS" ]; then + echo "profile covers only ${COVERED:-0} functions (expected >= ${MIN_FUNCTIONS})" >&2 + echo "the training workloads probably did not run; refusing to ship this profile" >&2 + exit 1 +fi +echo "Profile covers ${COVERED} functions." +end_step + +if [ "$TRAIN_ONLY" -eq 1 ]; then + echo + echo "To use the profile in a release build, add to RUSTFLAGS:" + echo " -Cprofile-use=${MERGED}" + echo "Total: $(fmt_duration $((SECONDS - SCRIPT_START)))" + exit 0 +fi + +begin_step "Step 5: optimized build" +cargo_build "$TARGET_DIR" "-Cprofile-use=${MERGED}" +end_step +echo +echo "Optimized binary: ${TARGET_DIR}/release/coreutils" +echo "Total: $(fmt_duration $((SECONDS - SCRIPT_START)))"