Files
wanderer/.github/workflows/test.yml
T
Dmitry Popov badd85902c Merge pull request #642 from guarzo/fix/test-failures-and-ci
fix(ci): make the test job fail when tests fail
2026-09-20 20:07:52 +02:00

440 lines
16 KiB
YAML

name: 🧪 Test Suite
on:
pull_request:
branches: [main, develop]
push:
branches: [main, develop]
# This workflow executes fork-authored code (mix deps.get / compile / test), so
# it holds no write scopes. The test-results comment is posted by the privileged
# pr-comment.yml workflow via workflow_run.
permissions:
contents: read
env:
MIX_ENV: test
ELIXIR_VERSION: '1.16'
OTP_VERSION: '26'
NODE_VERSION: '18'
jobs:
test:
name: Test Suite
runs-on: ubuntu-latest
services:
postgres:
image: postgres:15
env:
POSTGRES_PASSWORD: postgres
POSTGRES_DB: wanderer_test
options: >-
--health-cmd pg_isready
--health-interval 10s
--health-timeout 5s
--health-retries 5
ports:
- 5432:5432
steps:
- name: Checkout code
uses: actions/checkout@v7
- name: Setup Elixir/OTP
uses: erlef/setup-beam@v1
with:
elixir-version: ${{ env.ELIXIR_VERSION }}
otp-version: ${{ env.OTP_VERSION }}
- name: Cache Elixir dependencies
uses: actions/cache@v6
with:
path: |
deps
_build
key: ${{ runner.os }}-mix-${{ hashFiles('**/mix.lock') }}
restore-keys: ${{ runner.os }}-mix-
- name: Install Elixir dependencies
run: |
mix deps.get
mix deps.compile
- name: Check code formatting
id: format
run: |
if mix format --check-formatted; then
echo "status=✅ Passed" >> $GITHUB_OUTPUT
echo "count=0" >> $GITHUB_OUTPUT
else
echo "status=❌ Failed" >> $GITHUB_OUTPUT
echo "count=1" >> $GITHUB_OUTPUT
fi
continue-on-error: true
- name: Compile code and capture warnings
id: compile
run: |
# Capture compilation output
output=$(mix compile 2>&1 || true)
echo "$output" > compile_output.txt
# Count warnings
warning_count=$(echo "$output" | grep -c "warning:" || echo "0")
# Check if compilation succeeded
if mix compile > /dev/null 2>&1; then
echo "status=✅ Success" >> $GITHUB_OUTPUT
else
echo "status=❌ Failed" >> $GITHUB_OUTPUT
fi
echo "warnings=$warning_count" >> $GITHUB_OUTPUT
echo "output<<EOF" >> $GITHUB_OUTPUT
echo "$output" >> $GITHUB_OUTPUT
echo "EOF" >> $GITHUB_OUTPUT
continue-on-error: true
- name: Setup database
run: |
mix ecto.create
mix ecto.migrate
- name: Run tests with coverage
id: tests
run: |
# Run tests with coverage. The exit code is captured rather than
# discarded: this step must fail the job when the suite is red.
set +e
output=$(mix test --cover 2>&1)
test_exit_code=$?
set -e
echo "$output" > test_output.txt
# Parse test results
if echo "$output" | grep -q "0 failures"; then
echo "status=✅ All Passed" >> $GITHUB_OUTPUT
test_status="success"
else
echo "status=❌ Some Failed" >> $GITHUB_OUTPUT
test_status="failed"
fi
# Extract test counts
test_line=$(echo "$output" | grep -E "[0-9]+ tests?, [0-9]+ failures?" | head -1 || echo "0 tests, 0 failures")
total_tests=$(echo "$test_line" | grep -o '[0-9]\+ tests\?' | grep -o '[0-9]\+' | head -1 || echo "0")
failures=$(echo "$test_line" | grep -o '[0-9]\+ failures\?' | grep -o '[0-9]\+' | head -1 || echo "0")
echo "total=$total_tests" >> $GITHUB_OUTPUT
echo "failures=$failures" >> $GITHUB_OUTPUT
echo "passed=$((total_tests - failures))" >> $GITHUB_OUTPUT
# Calculate success rate
if [ "$total_tests" -gt 0 ]; then
success_rate=$(echo "scale=1; ($total_tests - $failures) * 100 / $total_tests" | bc)
else
success_rate="0"
fi
echo "success_rate=$success_rate" >> $GITHUB_OUTPUT
# Fail the job on a red suite. Reporting steps below still run
# because they depend on this step's outputs via always().
exit $test_exit_code
- name: Generate coverage report
if: always()
id: coverage
run: |
# Generate coverage report with GitHub format
output=$(mix coveralls.github 2>&1 || true)
echo "$output" > coverage_output.txt
# Extract coverage percentage
coverage=$(echo "$output" | grep -o '[0-9]\+\.[0-9]\+%' | head -1 | sed 's/%//' || echo "0")
if [ -z "$coverage" ]; then
coverage="0"
fi
echo "percentage=$coverage" >> $GITHUB_OUTPUT
# Determine status
if (( $(echo "$coverage >= 80" | bc -l) )); then
echo "status=✅ Excellent" >> $GITHUB_OUTPUT
elif (( $(echo "$coverage >= 60" | bc -l) )); then
echo "status=⚠️ Good" >> $GITHUB_OUTPUT
else
echo "status=❌ Needs Improvement" >> $GITHUB_OUTPUT
fi
continue-on-error: true
- name: Run Credo analysis
if: always()
id: credo
run: |
# Run Credo and capture output
output=$(mix credo --strict --format=json 2>&1 || true)
echo "$output" > credo_output.txt
# Try to parse JSON output
if echo "$output" | jq . > /dev/null 2>&1; then
issues=$(echo "$output" | jq '.issues | length' 2>/dev/null || echo "0")
high_issues=$(echo "$output" | jq '.issues | map(select(.priority == "high")) | length' 2>/dev/null || echo "0")
normal_issues=$(echo "$output" | jq '.issues | map(select(.priority == "normal")) | length' 2>/dev/null || echo "0")
low_issues=$(echo "$output" | jq '.issues | map(select(.priority == "low")) | length' 2>/dev/null || echo "0")
else
# Fallback: try to count issues from regular output
regular_output=$(mix credo --strict 2>&1 || true)
issues=$(echo "$regular_output" | grep -c "┃" || echo "0")
high_issues="0"
normal_issues="0"
low_issues="0"
fi
echo "total_issues=$issues" >> $GITHUB_OUTPUT
echo "high_issues=$high_issues" >> $GITHUB_OUTPUT
echo "normal_issues=$normal_issues" >> $GITHUB_OUTPUT
echo "low_issues=$low_issues" >> $GITHUB_OUTPUT
# Determine status
if [ "$issues" -eq 0 ]; then
echo "status=✅ Clean" >> $GITHUB_OUTPUT
elif [ "$issues" -lt 10 ]; then
echo "status=⚠️ Minor Issues" >> $GITHUB_OUTPUT
else
echo "status=❌ Needs Attention" >> $GITHUB_OUTPUT
fi
continue-on-error: true
- name: Run Dialyzer analysis
if: always()
id: dialyzer
run: |
# Ensure PLT is built
mix dialyzer --plt
# Run Dialyzer and capture output
output=$(mix dialyzer --format=github 2>&1 || true)
echo "$output" > dialyzer_output.txt
# Count warnings and errors
warnings=$(echo "$output" | grep -c "warning:" || echo "0")
errors=$(echo "$output" | grep -c "error:" || echo "0")
echo "warnings=$warnings" >> $GITHUB_OUTPUT
echo "errors=$errors" >> $GITHUB_OUTPUT
# Determine status
if [ "$errors" -eq 0 ] && [ "$warnings" -eq 0 ]; then
echo "status=✅ Clean" >> $GITHUB_OUTPUT
elif [ "$errors" -eq 0 ]; then
echo "status=⚠️ Warnings Only" >> $GITHUB_OUTPUT
else
echo "status=❌ Has Errors" >> $GITHUB_OUTPUT
fi
continue-on-error: true
- name: Create test results summary
if: always()
id: summary
run: |
# Calculate overall score
format_score=${{ steps.format.outputs.count == '0' && '100' || '0' }}
compile_score=${{ steps.compile.outputs.warnings == '0' && '100' || '80' }}
test_score=${{ steps.tests.outputs.success_rate }}
coverage_score=${{ steps.coverage.outputs.percentage }}
credo_score=$(echo "scale=0; (100 - ${{ steps.credo.outputs.total_issues }} * 2)" | bc | sed 's/^-.*$/0/')
dialyzer_score=$(echo "scale=0; (100 - ${{ steps.dialyzer.outputs.warnings }} * 2 - ${{ steps.dialyzer.outputs.errors }} * 10)" | bc | sed 's/^-.*$/0/')
overall_score=$(echo "scale=1; ($format_score + $compile_score + $test_score + $coverage_score + $credo_score + $dialyzer_score) / 6" | bc)
echo "overall_score=$overall_score" >> $GITHUB_OUTPUT
# Determine overall status
if (( $(echo "$overall_score >= 90" | bc -l) )); then
echo "overall_status=🌟 Excellent" >> $GITHUB_OUTPUT
elif (( $(echo "$overall_score >= 80" | bc -l) )); then
echo "overall_status=✅ Good" >> $GITHUB_OUTPUT
elif (( $(echo "$overall_score >= 70" | bc -l) )); then
echo "overall_status=⚠️ Needs Improvement" >> $GITHUB_OUTPUT
else
echo "overall_status=❌ Poor" >> $GITHUB_OUTPUT
fi
continue-on-error: true
# Fork PRs get a read-only GITHUB_TOKEN, so this workflow cannot post the
# comment itself. Instead the rendered body is uploaded as an artifact and
# posted by the privileged `pr-comment.yml` workflow via workflow_run.
- name: Render PR comment body
if: always() && github.event_name == 'pull_request'
env:
OVERALL_SCORE: ${{ steps.summary.outputs.overall_score }}
OVERALL_STATUS: ${{ steps.summary.outputs.overall_status }}
FORMAT_STATUS: ${{ steps.format.outputs.status }}
FORMAT_COUNT: ${{ steps.format.outputs.count }}
COMPILE_STATUS: ${{ steps.compile.outputs.status }}
COMPILE_WARNINGS: ${{ steps.compile.outputs.warnings }}
TESTS_STATUS: ${{ steps.tests.outputs.status }}
TESTS_FAILURES: ${{ steps.tests.outputs.failures }}
TESTS_TOTAL: ${{ steps.tests.outputs.total }}
TESTS_SUCCESS_RATE: ${{ steps.tests.outputs.success_rate }}
COVERAGE_STATUS: ${{ steps.coverage.outputs.status }}
COVERAGE_PERCENTAGE: ${{ steps.coverage.outputs.percentage }}
CREDO_STATUS: ${{ steps.credo.outputs.status }}
CREDO_TOTAL: ${{ steps.credo.outputs.total_issues }}
CREDO_HIGH: ${{ steps.credo.outputs.high_issues }}
CREDO_NORMAL: ${{ steps.credo.outputs.normal_issues }}
CREDO_LOW: ${{ steps.credo.outputs.low_issues }}
DIALYZER_STATUS: ${{ steps.dialyzer.outputs.status }}
DIALYZER_ERRORS: ${{ steps.dialyzer.outputs.errors }}
DIALYZER_WARNINGS: ${{ steps.dialyzer.outputs.warnings }}
PR_NUMBER: ${{ github.event.pull_request.number }}
run: |
mkdir -p pr-comment
# Validate the PR number so the trusted workflow can trust this file.
case "$PR_NUMBER" in
''|*[!0-9]*) echo "Refusing to emit non-numeric PR number" >&2; exit 1 ;;
esac
printf '%s\n' "$PR_NUMBER" > pr-comment/pr-number
# Values are interpolated by the shell from env, never by Actions
# expression substitution, so tool output cannot inject workflow syntax.
cat > pr-comment/body.md <<EOF
## 🧪 Test Results Summary
**Overall Quality Score: ${OVERALL_SCORE}%** ${OVERALL_STATUS}
### 📊 Metrics Dashboard
| Category | Status | Count | Details |
|----------|---------|-------|---------|
| 📝 **Code Formatting** | ${FORMAT_STATUS} | ${FORMAT_COUNT} issues | \`mix format --check-formatted\` |
| 🔨 **Compilation** | ${COMPILE_STATUS} | ${COMPILE_WARNINGS} warnings | \`mix compile\` |
| 🧪 **Tests** | ${TESTS_STATUS} | ${TESTS_FAILURES}/${TESTS_TOTAL} failed | Success rate: ${TESTS_SUCCESS_RATE}% |
| 📊 **Coverage** | ${COVERAGE_STATUS} | ${COVERAGE_PERCENTAGE}% | \`mix coveralls\` |
| 🎯 **Credo** | ${CREDO_STATUS} | ${CREDO_TOTAL} issues | High: ${CREDO_HIGH}, Normal: ${CREDO_NORMAL}, Low: ${CREDO_LOW} |
| 🔍 **Dialyzer** | ${DIALYZER_STATUS} | ${DIALYZER_ERRORS} errors, ${DIALYZER_WARNINGS} warnings | \`mix dialyzer\` |
### 🎯 Quality Gates
Based on the project's quality thresholds:
- **Compilation Warnings**: ${COMPILE_WARNINGS}/148 (limit: 148)
- **Credo Issues**: ${CREDO_TOTAL}/87 (limit: 87)
- **Dialyzer Warnings**: ${DIALYZER_WARNINGS}/161 (limit: 161)
- **Test Coverage**: ${COVERAGE_PERCENTAGE}%/50% (minimum: 50%)
- **Test Failures**: ${TESTS_FAILURES}/0 (limit: 0)
<details>
<summary>📈 Progress Toward Goals</summary>
Target goals for the project:
- ✨ **Zero compilation warnings** (currently: ${COMPILE_WARNINGS})
- ✨ **≤10 Credo issues** (currently: ${CREDO_TOTAL})
- ✨ **Zero Dialyzer warnings** (currently: ${DIALYZER_WARNINGS})
- ✨ **≥85% test coverage** (currently: ${COVERAGE_PERCENTAGE}%)
- ✅ **Zero test failures** (currently: ${TESTS_FAILURES})
</details>
<details>
<summary>🔧 Quick Actions</summary>
To improve code quality:
\`\`\`bash
# Fix formatting issues
mix format
# View detailed Credo analysis
mix credo --strict
# Check Dialyzer warnings
mix dialyzer
# Generate detailed coverage report
mix coveralls.html
\`\`\`
</details>
---
🤖 *Auto-generated by GitHub Actions* • Updated: $(date -u '+%Y-%m-%d %H:%M UTC')
> **Note**: This comment will be updated automatically when new commits are pushed to this PR.
EOF
# Mirror to the run summary so the report is visible even if the
# follow-up commenting workflow is unavailable.
cat pr-comment/body.md >> "$GITHUB_STEP_SUMMARY"
continue-on-error: true
- name: Upload PR comment body
if: always() && github.event_name == 'pull_request'
uses: actions/upload-artifact@v7
with:
name: pr-comment
path: pr-comment/
retention-days: 1
continue-on-error: true
# Integration tests are excluded from the default `mix test` run by
# test/test_helper.exs, so until now they had never executed in CI at all.
# They run here as a separate job rather than by changing the default
# exclusion, so local `mix test` stays fast.
#
# This is a hard gate: a red integration run blocks the merge. Note that 30 of
# the tests `--only integration` selects are `@tag :skip`, so the gate covers
# the 35 that actually execute.
integration-test:
name: Integration Tests
runs-on: ubuntu-latest
services:
postgres:
image: postgres:15
env:
POSTGRES_PASSWORD: postgres
POSTGRES_DB: wanderer_test
options: >-
--health-cmd pg_isready
--health-interval 10s
--health-timeout 5s
--health-retries 5
ports:
- 5432:5432
steps:
- name: Checkout code
uses: actions/checkout@v4
- name: Setup Elixir/OTP
uses: erlef/setup-beam@v1
with:
elixir-version: ${{ env.ELIXIR_VERSION }}
otp-version: ${{ env.OTP_VERSION }}
- name: Cache Elixir dependencies
uses: actions/cache@v3
with:
path: |
deps
_build
key: ${{ runner.os }}-mix-${{ hashFiles('**/mix.lock') }}
restore-keys: ${{ runner.os }}-mix-
- name: Install Elixir dependencies
run: |
mix deps.get
mix deps.compile
- name: Setup database
run: |
mix ecto.create
mix ecto.migrate
- name: Run integration tests
# --only, not --include: --include would run the whole suite on top of
# the tagged tests, duplicating the main test job's ~10 minutes and
# re-reporting failures it already covers.
run: mix test --only integration