Repository navigation
ci(benchmark): skip PR comment on fork PRs, make headline reflect rea… #133
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| name: Performance Regression | |
| # ───────────────────────────────────────────────────────────────────────────── | |
| # Two jobs: | |
| # | |
| # layout-regression — runs on every PR and every push to main. | |
| # • Executes benchmark/layout_regression.dart | |
| # • Fails if any fixture's median layout time exceeds its hard threshold | |
| # • Posts a PR comment with the full results table | |
| # | |
| # full-benchmark — runs weekly + on release branches. | |
| # • Executes the original benchmark/parse_benchmark.dart (throughput info) | |
| # • Never fails CI (informational only) — results are uploaded as artifacts | |
| # ───────────────────────────────────────────────────────────────────────────── | |
| env: | |
| FLUTTER_VERSION: "3.41.5" # keep in sync with golden.yml | |
| on: | |
| pull_request: | |
| branches: [main, develop] | |
| push: | |
| branches: [main] | |
| schedule: | |
| - cron: '0 0 * * 0' # weekly full benchmark (Sunday midnight UTC) | |
| workflow_dispatch: | |
| jobs: | |
| # ── Layout regression guard (runs on every PR) ──────────────────────────── | |
| layout-regression: | |
| name: Layout Regression (60 FPS guard) | |
| runs-on: ubuntu-22.04 | |
| # Skip on the weekly schedule — that's for the full-benchmark job only | |
| if: github.event_name != 'schedule' | |
| permissions: | |
| pull-requests: write | |
| steps: | |
| - name: Checkout | |
| uses: actions/checkout@v4 | |
| - name: Setup Flutter (pinned) | |
| uses: subosito/flutter-action@v2 | |
| with: | |
| flutter-version: ${{ env.FLUTTER_VERSION }} | |
| channel: stable | |
| cache: true | |
| - name: Get dependencies | |
| run: flutter pub get | |
| - name: Run layout regression benchmark | |
| id: bench | |
| run: | | |
| mkdir -p benchmark/results | |
| # Run with JSON reporter so we can parse pass/fail. | |
| # || true — CI runners use software rendering and are 2-3x slower than | |
| # real hardware (iPhone 13 / Pixel 6). Thresholds are calibrated for | |
| # real devices. Keep results for trend tracking but never block CI. | |
| flutter test benchmark/layout_regression.dart \ | |
| --reporter expanded \ | |
| 2>&1 | tee benchmark/results/ci_run.txt || true | |
| # Always treat as passed from CI's perspective | |
| EXIT_CODE=0 | |
| # Parse the JSON result files for the summary table | |
| SUMMARY=$(python3 - <<'PYEOF' | |
| import json, os, glob | |
| files = sorted(glob.glob('benchmark/results/layout_*.json')) | |
| if not files: | |
| print("No result file generated.") | |
| else: | |
| data = json.load(open(files[-1])) | |
| rows = [] | |
| any_fail = False | |
| for r in data.get('results', []): | |
| icon = "✅" if r["passed"] else "❌" | |
| rows.append( | |
| f"| {icon} | `{r['fixture']}` | {r['threshold_ms']} | " | |
| f"{r['median_ms']} | {r['p95_ms']} |" | |
| ) | |
| if not r["passed"]: | |
| any_fail = True | |
| header = ( | |
| "| | Fixture | Budget (ms) | Median (ms) | P95 (ms) |\n" | |
| "|---|---|---|---|---|" | |
| ) | |
| print(header) | |
| print("\n".join(rows)) | |
| if any_fail: | |
| print("\n**One or more fixtures exceeded the 16 ms budget.**") | |
| PYEOF | |
| ) | |
| echo "summary<<EOF" >> "$GITHUB_OUTPUT" | |
| echo "$SUMMARY" >> "$GITHUB_OUTPUT" | |
| echo "EOF" >> "$GITHUB_OUTPUT" | |
| echo "exit_code=$EXIT_CODE" >> "$GITHUB_OUTPUT" | |
| # The job never fails (see above), so the comment headline must come | |
| # from the actual results, not from EXIT_CODE. | |
| if echo "$SUMMARY" | grep -q "exceeded"; then | |
| echo "budget_ok=false" >> "$GITHUB_OUTPUT" | |
| else | |
| echo "budget_ok=true" >> "$GITHUB_OUTPUT" | |
| fi | |
| exit $EXIT_CODE | |
| - name: Upload result JSON | |
| if: always() | |
| uses: actions/upload-artifact@v4 | |
| with: | |
| name: layout-regression-${{ github.run_number }} | |
| path: benchmark/results/ | |
| retention-days: 30 | |
| # Fork PRs get a read-only GITHUB_TOKEN, so createComment would fail with | |
| # "Resource not accessible by integration" and turn the whole job red. | |
| # Skip them; results are still in the uploaded artifact. continue-on-error | |
| # is a second guard so a comment hiccup never fails the check. | |
| - name: Post PR comment | |
| if: >- | |
| always() && github.event_name == 'pull_request' && | |
| github.event.pull_request.head.repo.full_name == github.repository | |
| continue-on-error: true | |
| uses: actions/github-script@v7 | |
| env: | |
| BENCH_BUDGET_OK: ${{ steps.bench.outputs.budget_ok }} | |
| BENCH_SUMMARY: ${{ steps.bench.outputs.summary }} | |
| with: | |
| script: | | |
| const summary = process.env.BENCH_SUMMARY || ''; | |
| const passed = process.env.BENCH_BUDGET_OK === 'true'; | |
| const headline = passed | |
| ? '## ✅ Layout Regression — All fixtures within 60 FPS budget' | |
| : '## ⚠️ Layout Regression — Budget exceeded (informational)'; | |
| const body = [ | |
| headline, | |
| '', | |
| summary, | |
| '', | |
| `> Flutter \`${{ env.FLUTTER_VERSION }}\` · ubuntu-22.04`, | |
| '', | |
| passed | |
| ? '_No action required._' | |
| : [ | |
| '**Informational, does not block merge:** CI runners use software', | |
| 'rendering and are 2-3x slower than the devices the budgets are', | |
| 'calibrated for. Compare against the base branch; if a fixture', | |
| 'regressed relative to it, profile with:', | |
| '```bash', | |
| 'flutter test benchmark/layout_regression.dart --reporter expanded', | |
| '```', | |
| 'and check `_performLineLayout` / `_buildCharacterMapping` for', | |
| 'any new O(N²) or O(N log N) paths introduced in this PR.', | |
| ].join('\n'), | |
| ].join('\n'); | |
| await github.rest.issues.createComment({ | |
| owner: context.repo.owner, | |
| repo: context.repo.repo, | |
| issue_number: context.issue.number, | |
| body, | |
| }); | |
| # ── Full throughput benchmark (weekly, informational) ──────────────────── | |
| full-benchmark: | |
| name: Full Throughput Benchmark | |
| runs-on: ubuntu-22.04 | |
| if: >- | |
| github.event_name == 'schedule' || | |
| github.event_name == 'workflow_dispatch' || | |
| startsWith(github.ref, 'refs/heads/release/') | |
| # Needed by the failure-alert step below to open a tracking issue. | |
| permissions: | |
| issues: write | |
| steps: | |
| - name: Checkout | |
| uses: actions/checkout@v4 | |
| - name: Setup Flutter (pinned) | |
| uses: subosito/flutter-action@v2 | |
| with: | |
| flutter-version: ${{ env.FLUTTER_VERSION }} | |
| channel: stable | |
| cache: true | |
| - name: Get dependencies | |
| run: flutter pub get | |
| - name: Run parse benchmarks | |
| run: | | |
| # Must exist before `tee` writes into it — this job does not share a | |
| # filesystem with the PR job that also creates it. Without this the | |
| # step fails with "tee: ...: No such file or directory". | |
| mkdir -p benchmark/results | |
| flutter test benchmark/parse_benchmark.dart \ | |
| --no-test-randomize-ordering-seed \ | |
| --reporter expanded \ | |
| 2>&1 | tee benchmark/results/parse_$(date +%Y%m%d).txt | |
| - name: Run layout regression (informational — never fails here) | |
| run: | | |
| flutter test benchmark/layout_regression.dart \ | |
| --reporter expanded \ | |
| 2>&1 | tee benchmark/results/layout_$(date +%Y%m%d).txt || true | |
| - name: Upload benchmark results | |
| uses: actions/upload-artifact@v4 | |
| with: | |
| name: benchmark-full-${{ github.run_number }} | |
| path: benchmark/results/ | |
| retention-days: 90 | |
| # A scheduled job that fails is otherwise invisible — it just shows a red | |
| # X in the Actions tab that nobody watches (this is how a `mkdir` typo | |
| # silently broke the weekly benchmark for a month). Open/refresh a | |
| # tracking issue so the failure surfaces where it will be seen. | |
| - name: Alert on scheduled-run failure | |
| if: failure() && github.event_name == 'schedule' | |
| uses: actions/github-script@v7 | |
| with: | |
| script: | | |
| const title = '⚠️ Weekly benchmark workflow failed'; | |
| const runUrl = `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}`; | |
| const body = [ | |
| `The scheduled **Full Throughput Benchmark** run failed.`, | |
| ``, | |
| `- Run: ${runUrl}`, | |
| `- Commit: ${context.sha}`, | |
| ``, | |
| `This issue was opened automatically so a scheduled failure is not`, | |
| `silently ignored. Close it once the run is green again.`, | |
| ].join('\n'); | |
| // Reuse an existing open alert issue instead of spamming a new one | |
| // each week. | |
| const existing = await github.rest.issues.listForRepo({ | |
| owner: context.repo.owner, | |
| repo: context.repo.repo, | |
| state: 'open', | |
| labels: 'ci-benchmark-failure', | |
| }); | |
| if (existing.data.length > 0) { | |
| await github.rest.issues.createComment({ | |
| owner: context.repo.owner, | |
| repo: context.repo.repo, | |
| issue_number: existing.data[0].number, | |
| body, | |
| }); | |
| } else { | |
| await github.rest.issues.create({ | |
| owner: context.repo.owner, | |
| repo: context.repo.repo, | |
| title, | |
| body, | |
| labels: ['ci-benchmark-failure'], | |
| }); | |
| } |