mirror of
https://github.com/rustfs/rustfs.git
synced 2026-09-03 06:37:34 +08:00
The repository_dispatch handoff step was continue-on-error with a single attempt: if the call failed (token lacking contents:write, transient API error), the chain stalled silently while every job stayed green. Each handoff now retries 3x and, if all attempts fail, files an alert issue in rustfs/backlog with the exact recovery command before exiting 1 (still continue-on-error, so suite workflows themselves never fail).
340 lines
14 KiB
YAML
340 lines
14 KiB
YAML
name: RustFS Heal Test
|
|
|
|
on:
|
|
workflow_dispatch:
|
|
inputs:
|
|
package_url:
|
|
description: 'Direct .deb URL (nightly/R2). Defaults to the latest nightly deb.'
|
|
required: false
|
|
type: string
|
|
stop_node_gb:
|
|
description: 'Stop the outage node when surviving nodes reach N GiB'
|
|
required: false
|
|
default: '15'
|
|
warp_stop_gb:
|
|
description: 'Stop warp when surviving nodes reach N GiB'
|
|
required: false
|
|
default: '40'
|
|
cleanup_before:
|
|
description: 'Reset the nodes before the test (DESTROYS existing data/config)'
|
|
type: boolean
|
|
default: true
|
|
cleanup_after:
|
|
description: 'Reset the nodes after the test (DESTROYS test data/config)'
|
|
type: boolean
|
|
default: true
|
|
repository_dispatch:
|
|
# Chain handoff: dispatched when the storage suite finishes. Heal runs
|
|
# exactly once per chain; the pool expansion workflow no longer embeds
|
|
# its own heal pass.
|
|
types: [rustfs-chain-heal]
|
|
|
|
permissions:
|
|
contents: read
|
|
|
|
# Only one test at a time: both this and the pool-expansion workflow mutate
|
|
# the same test environment, so they share one concurrency group.
|
|
concurrency:
|
|
group: rustfs-shared-functional-tests
|
|
cancel-in-progress: false
|
|
|
|
defaults:
|
|
run:
|
|
shell: bash
|
|
|
|
env:
|
|
RUSTFS_ACCESS_KEY: ${{ secrets.RUSTFS_ACCESS_KEY }}
|
|
RUSTFS_SECRET_KEY: ${{ secrets.RUSTFS_SECRET_KEY }}
|
|
RUSTFS_API_ENDPOINT: ${{ secrets.RUSTFS_API_ENDPOINT || vars.RUSTFS_API_ENDPOINT || vars.RUSTFS_RC_ENDPOINT }}
|
|
RUSTFS_NODES: ${{ secrets.RUSTFS_NODES || vars.RUSTFS_NODES }}
|
|
RUSTFS_SSH_USER: ${{ secrets.RUSTFS_SSH_USER || vars.RUSTFS_SSH_USER }}
|
|
PF_TESTING_GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
|
RUSTFS_NIGHTLY_PACKAGE_URL: ${{ vars.RUSTFS_NIGHTLY_PACKAGE_URL || 'https://dl.rustfs.com/artifacts/rustfs/packages/nightly/rustfs-nightly-latest.deb' }}
|
|
|
|
jobs:
|
|
heal-test:
|
|
runs-on: smoke-testing
|
|
# Requirement: a failing suite must not fail the workflow; failures
|
|
# are filed to rustfs/backlog and the chain continues.
|
|
continue-on-error: true
|
|
timeout-minutes: 480
|
|
# Standalone manual run, or one link of the nightly functional chain
|
|
# (storage -> heal -> pool). Pool expansion no longer re-runs heal.
|
|
if: ${{ github.event_name == 'workflow_dispatch' || github.event_name == 'repository_dispatch' }}
|
|
steps:
|
|
# auto-testing is private: clone it with the dedicated PF token (not
|
|
# GITHUB_TOKEN) and retry transient GitHub/network failures.
|
|
- name: Checkout auto-testing scripts (with retry)
|
|
env:
|
|
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
|
run: |
|
|
set -euo pipefail
|
|
rm -rf auto-testing
|
|
for attempt in 1 2 3 4 5; do
|
|
if gh repo clone rustfs/auto-testing auto-testing -- --depth 1 --quiet; then
|
|
echo "auto-testing cloned (attempt ${attempt})"
|
|
exit 0
|
|
fi
|
|
rm -rf auto-testing
|
|
echo "clone attempt ${attempt} failed; retrying in $((attempt * 15))s" >&2
|
|
sleep $((attempt * 15))
|
|
done
|
|
echo "ERROR: unable to clone rustfs/auto-testing after 5 attempts" >&2
|
|
exit 1
|
|
|
|
- name: Show environment
|
|
run: |
|
|
uname -a
|
|
jq --version
|
|
openssl version
|
|
warp --version || true
|
|
df -h /data | tail -1
|
|
|
|
- name: Cleanup environment (before)
|
|
if: ${{ inputs.cleanup_before != 'false' }}
|
|
run: |
|
|
set -euo pipefail
|
|
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
|
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
|
for node in "${NODES[@]}"; do
|
|
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
|
set -euo pipefail
|
|
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
|
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
|
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
|
${SUDO} dpkg -P rustfs
|
|
fi
|
|
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
|
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
|
'
|
|
done
|
|
|
|
- name: Install RustFS package & start cluster
|
|
run: |
|
|
ARGS=(--steps "1,2" -y --endpoint "${{ env.RUSTFS_API_ENDPOINT }}")
|
|
if [ -n "${{ inputs.package_url }}" ]; then
|
|
ARGS+=(--package-url "${{ inputs.package_url }}")
|
|
else
|
|
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
|
|
fi
|
|
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
|
|
|
|
- name: Preflight checks
|
|
run: |
|
|
ARGS=(--preflight --endpoint "${{ env.RUSTFS_API_ENDPOINT }}")
|
|
if [ -n "${{ inputs.package_url }}" ]; then
|
|
ARGS+=(--package-url "${{ inputs.package_url }}")
|
|
else
|
|
ARGS+=(--package-url "${{ env.RUSTFS_NIGHTLY_PACKAGE_URL }}")
|
|
fi
|
|
./auto-testing/rustfs_heal_test.sh "${ARGS[@]}"
|
|
|
|
- name: Run heal test (write -> outage -> heal -> verify)
|
|
id: test
|
|
run: |
|
|
./auto-testing/rustfs_heal_test.sh \
|
|
--steps "3,4,5,6,7" -y \
|
|
--endpoint "${{ env.RUSTFS_API_ENDPOINT }}" \
|
|
--stop-node-gb "${{ inputs.stop_node_gb || '15' }}" \
|
|
--warp-stop-gb "${{ inputs.warp_stop_gb || '40' }}" \
|
|
--log-file /tmp/rustfs-heal-test.log
|
|
|
|
- name: Generate report
|
|
if: always()
|
|
env:
|
|
LOG_FILE: /tmp/rustfs-heal-test.log
|
|
REPORT_FILE: /tmp/rustfs-heal-report.md
|
|
run: |
|
|
set -euo pipefail
|
|
PACKAGE_URL='${{ inputs.package_url }}'
|
|
if [ -n "${PACKAGE_URL}" ]; then
|
|
PACKAGE_SOURCE="${PACKAGE_URL}"
|
|
else
|
|
PACKAGE_SOURCE="${RUSTFS_NIGHTLY_PACKAGE_URL}"
|
|
fi
|
|
{
|
|
echo "# RustFS heal test report"
|
|
echo ""
|
|
echo "- Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}"
|
|
echo "- Trigger: ${{ github.event_name }}"
|
|
echo "- Package: ${PACKAGE_SOURCE}"
|
|
echo "- Test Step Outcome: ${{ steps.test.outcome }}"
|
|
echo ""
|
|
echo "## Log tail"
|
|
echo '```text'
|
|
tail -n 200 "${LOG_FILE}" || true
|
|
echo '```'
|
|
} | tee "${REPORT_FILE}"
|
|
cat "${REPORT_FILE}" >> "${GITHUB_STEP_SUMMARY}"
|
|
|
|
- name: Upload functional report to dashboard
|
|
if: always()
|
|
continue-on-error: true
|
|
env:
|
|
GH_TOKEN: ${{ env.PF_TESTING_GH_TOKEN }}
|
|
REPORT_FILE: /tmp/rustfs-heal-report.md
|
|
SUITE: heal
|
|
run: |
|
|
set -euo pipefail
|
|
if [ -z "${GH_TOKEN:-}" ]; then
|
|
echo "PF_TESTING_GH_TOKEN is not configured; skipping dashboard upload"
|
|
exit 0
|
|
fi
|
|
DATE="$(date -u +%Y-%m-%d)"
|
|
REPORT_PATH="functional-reports/${SUITE}/${DATE}.md"
|
|
CONTENT="$(python3 -c 'import base64,sys;print(base64.b64encode(open(sys.argv[1],"rb").read()).decode())' "${REPORT_FILE}")"
|
|
SHA="$(gh api "repos/rustfs/dashboard/contents/${REPORT_PATH}" -q '.sha' 2>/dev/null || true)"
|
|
if [ -n "${SHA}" ]; then
|
|
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" --arg sha "${SHA}" \
|
|
'{message:$msg, content:$content, sha:$sha}' \
|
|
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
|
else
|
|
jq -n --arg msg "report(${SUITE}): ${DATE}" --arg content "${CONTENT}" \
|
|
'{message:$msg, content:$content}' \
|
|
| gh api --method PUT "repos/rustfs/dashboard/contents/${REPORT_PATH}" --input - >/dev/null
|
|
fi
|
|
|
|
- name: File failure issue in rustfs/backlog
|
|
if: ${{ always() && (failure() || steps.test.outcome == 'failure' || steps.test.outcome == 'cancelled') }}
|
|
continue-on-error: true
|
|
env:
|
|
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
|
SUITE: 'heal'
|
|
SUITE_LABEL: 'Heal'
|
|
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
|
|
REPORT_FILE: '/tmp/rustfs-heal-report.md'
|
|
LOG_FILE: '/tmp/rustfs-heal-test.log'
|
|
run: |
|
|
set -euo pipefail
|
|
if [ -z "${GH_TOKEN:-}" ]; then
|
|
echo "PF_TESTING_GH_TOKEN is not configured; skipping backlog issue"
|
|
exit 0
|
|
fi
|
|
TITLE="[functional][${SUITE}] ${SUITE_LABEL} suite failed (run ${GITHUB_RUN_ID})"
|
|
EXISTING="$(gh issue list -R rustfs/backlog --state all \
|
|
--search "in:title \"run ${GITHUB_RUN_ID}\"" \
|
|
--json number --jq '.[].number' || true)"
|
|
if [ -n "${EXISTING}" ]; then
|
|
echo "backlog issue already exists for run ${GITHUB_RUN_ID}; skipping"
|
|
exit 0
|
|
fi
|
|
redact() {
|
|
sed -E \
|
|
-e 's/(RUSTFS_(ACCESS_KEY|SECRET_KEY)[=: ]+)[^[:space:]]+/\1[REDACTED]/Ig' \
|
|
-e 's/(Authorization:).*/\1 [REDACTED]/Ig' \
|
|
-e 's/(X-Amz-Signature=)[^&[:space:]]+/\1[REDACTED]/Ig' \
|
|
-e 's/^.*(password|secret|token)[=: ].*/[REDACTED SENSITIVE LINE]/Ig'
|
|
}
|
|
BODY_FILE="$(mktemp)"
|
|
{
|
|
echo "The **${SUITE_LABEL}** functional suite failed."
|
|
echo ""
|
|
echo "- Suite: \`${SUITE}\`"
|
|
echo "- Run: ${RUN_URL}"
|
|
echo "- Trigger: ${GITHUB_EVENT_NAME}"
|
|
echo "- Date: $(date -u +%Y-%m-%d)"
|
|
echo ""
|
|
echo "## Report (errors and symptoms)"
|
|
echo ""
|
|
if [ -s "${REPORT_FILE}" ]; then
|
|
redact < "${REPORT_FILE}"
|
|
elif [ -s "${LOG_FILE:-}" ]; then
|
|
echo "(report file missing; log tail below)"
|
|
echo ""
|
|
tail -n 200 "${LOG_FILE}" | redact
|
|
else
|
|
echo "(no report or log file was produced)"
|
|
fi
|
|
} | head -c 55000 > "${BODY_FILE}"
|
|
gh label create functional-test -R rustfs/backlog --color d73a4a 2>/dev/null || true
|
|
if ! gh issue create -R rustfs/backlog --title "${TITLE}" \
|
|
--body-file "${BODY_FILE}" --label functional-test; then
|
|
gh issue create -R rustfs/backlog --title "${TITLE}" --body-file "${BODY_FILE}"
|
|
fi
|
|
echo "filed backlog issue for suite ${SUITE}"
|
|
|
|
- name: Upload test logs
|
|
if: always()
|
|
uses: actions/upload-artifact@b7c566a772e6b6bfb58ed0dc250532a479d7789f # v6
|
|
with:
|
|
name: rustfs-heal-test-${{ github.run_id }}
|
|
path: |
|
|
/tmp/rustfs-heal-test*.log
|
|
/tmp/rustfs-warp.*.log
|
|
if-no-files-found: warn
|
|
|
|
- name: Cleanup environment (after)
|
|
if: ${{ always() && inputs.cleanup_after != 'false' }}
|
|
run: |
|
|
set -euo pipefail
|
|
read -r -a NODES <<< "${RUSTFS_NODES:-vm000 vm001 vm002}"
|
|
SSH_USER="${RUSTFS_SSH_USER:-azureuser}"
|
|
for node in "${NODES[@]}"; do
|
|
ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${SSH_USER}@${node}" '
|
|
set -euo pipefail
|
|
SUDO=""; [ "$(id -u)" -ne 0 ] && SUDO="sudo -n"
|
|
${SUDO} systemctl stop rustfs 2>/dev/null || true
|
|
if ${SUDO} dpkg -l rustfs 2>/dev/null | grep -q "^ii"; then
|
|
${SUDO} dpkg -P rustfs
|
|
fi
|
|
for i in 1 2 3 4; do ${SUDO} rm -rf /data/rustfs${i}/mnmd; done
|
|
${SUDO} rm -rf /var/log/rustfs /var/lib/rustfs/kms /var/lib/rustfs/kms-backup
|
|
'
|
|
done
|
|
|
|
- name: "Continue functional chain (next: Pool expansion)"
|
|
# Only chain-triggered runs forward to the next suite; standalone
|
|
# workflow_dispatch runs stop after their own cleanup. A failed
|
|
# handoff must never pass silently: it retries, then files an alert
|
|
# issue in rustfs/backlog so a stalled chain is visible.
|
|
if: ${{ always() && github.event_name == 'repository_dispatch' }}
|
|
continue-on-error: true
|
|
env:
|
|
GH_TOKEN: ${{ secrets.PF_TESTING_GH_TOKEN }}
|
|
run: |
|
|
set -uo pipefail
|
|
if [ -z "${{GH_TOKEN:-}}" ]; then
|
|
echo "PF_TESTING_GH_TOKEN is not configured; cannot dispatch the next suite" >&2
|
|
exit 1
|
|
fi
|
|
DISPATCHED=0
|
|
for attempt in 1 2 3; do
|
|
if gh api --method POST repos/rustfs/rustfs/dispatches \
|
|
-f event_type='rustfs-chain-pool' \
|
|
-F 'client_payload[from_suite]=heal'; then
|
|
echo "dispatched next suite Pool expansion (attempt ${{attempt}})"
|
|
DISPATCHED=1
|
|
break
|
|
fi
|
|
echo "dispatch attempt ${{attempt}} failed; retrying in ${{attempt}}0s" >&2
|
|
sleep "${{attempt}}0"
|
|
done
|
|
if [ "${{DISPATCHED:-0}}" -ne 1 ]; then
|
|
echo "ERROR: functional chain stalled: could not dispatch Pool expansion after 3 attempts" >&2
|
|
TITLE="[functional][chain] stalled after heal (run ${{GITHUB_RUN_ID}})"
|
|
BODY_FILE="$(mktemp)"
|
|
{
|
|
echo "The functional chain could not hand off from **heal** to **Pool expansion** after 3 attempts."
|
|
echo ""
|
|
echo "- Failed suite job: ${{GITHUB_SERVER_URL}}/${{GITHUB_REPOSITORY}}/actions/runs/${{GITHUB_RUN_ID}}"
|
|
echo "- Expected next event: `rustfs-chain-pool`"
|
|
echo "- Likely cause: PF_TESTING_GH_TOKEN lacks contents:write on rustfs/rustfs, or the GitHub API was unavailable."
|
|
echo "- Recovery: re-dispatch manually with"
|
|
echo " ```"
|
|
echo " gh api --method POST repos/rustfs/rustfs/dispatches -f event_type='rustfs-chain-pool'"
|
|
echo " ```"
|
|
} > "${{BODY_FILE}}"
|
|
gh issue create -R rustfs/backlog --title "${{TITLE}}" \
|
|
--body-file "${{BODY_FILE}}" --label functional-test \
|
|
|| gh issue create -R rustfs/backlog --title "${{TITLE}}" --body-file "${{BODY_FILE}}" \
|
|
|| echo "could not file the stall alert issue either; check the token" >&2
|
|
exit 1
|
|
fi
|
|
|
|
- name: Notify on failure
|
|
if: failure()
|
|
run: |
|
|
echo "RustFS heal test failed"
|
|
echo "Package source: ${{ inputs.package_url || 'nightly (R2 latest)' }}"
|
|
echo "See the uploaded log artifact for details."
|