traffic-health #4461
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # Hand-maintained (NOT generated by Upptime; update-template leaves it alone). | |
| # | |
| # Closes checker incidents that the checker already recovered but never closed. | |
| # | |
| # Why this exists: upptime/uptime-monitor commits history/<slug>.yml and then | |
| # closes the incident issue in the SAME run. Its push does not rebase, so when | |
| # another workflow's commit landed on main first the run died with | |
| # "! [rejected] main -> main (fetch first)" AFTER the history file already | |
| # said `status: up` locally; the next run re-pushed the history but, seeing | |
| # the component up, skipped the incident step, and the issue stayed open until | |
| # the NEXT transition. Issue #13 (2026-09-12) stayed open 17 h 26 min for a | |
| # ~30-minute degradation that way, and Upptime counts open-to-close as | |
| # downtime, so the API's uptime figures for two days were wrong. | |
| # | |
| # The push race itself is fixed by putting every workflow that pushes to main | |
| # in Upptime's concurrency group (traffic-health.yml, server-latency.yml). | |
| # This job is the safety net: for every OPEN `status`-labelled issue opened by | |
| # the checker, read the component's history file from main; if it reads `up` | |
| # and its lastUpdated is more than 10 minutes after the issue was opened, the | |
| # recovery run failed before closing the issue, so close it here. | |
| # | |
| # Scope: checker-authored issues only (body matches Upptime's template). | |
| # Manual incidents (a human opens an issue with the `status` label, README | |
| # "Manual incident via issue") and `traffic-alert` issues (owned by | |
| # traffic-health.yml) are never touched. Closes with github.token, so the | |
| # comment is authored by github-actions[bot], not by the GH_PAT owner. | |
| name: Stale Incidents | |
| on: | |
| # Rides the traffic-health dispatch the Vercel cron fires every 5 minutes | |
| # (two minutes after the checker), so it runs right after a checker run | |
| # would have closed the issue. GitHub's schedule is a fallback only. | |
| repository_dispatch: | |
| types: [traffic-health] | |
| schedule: | |
| - cron: "*/15 * * * *" | |
| workflow_dispatch: | |
| # Upptime's group: serializes this against the checker so the two never act | |
| # on the same incident at once. Never cancel-in-progress in this group. | |
| # Own group: this workflow only closes issues and never pushes to main, so it | |
| # must not compete with the checker's push group (a pending run there is | |
| # cancelled by the next arrival; two runs were cancelled that way on 2026-09-13). | |
| concurrency: | |
| group: ${{ github.repository }}-stale-incidents | |
| cancel-in-progress: false | |
| permissions: | |
| contents: read | |
| issues: write | |
| env: | |
| # Minutes the checker must have read `up` past the issue's created_at before | |
| # the issue is treated as stranded rather than freshly opened. | |
| GRACE_MINUTES: "10" | |
| jobs: | |
| reconcile: | |
| name: Close incidents the checker recovered but left open | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 5 | |
| steps: | |
| - name: Checkout | |
| uses: actions/checkout@v6 | |
| with: | |
| # Always read the latest history from main, whatever ref fired us. | |
| ref: main | |
| - name: Reconcile open status issues against history/ | |
| env: | |
| GH_TOKEN: ${{ github.token }} | |
| REPO: ${{ github.repository }} | |
| RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} | |
| run: | | |
| set -euo pipefail | |
| gh issue list --repo "$REPO" --label status --state open --limit 100 \ | |
| --json number,title,createdAt,labels,body,url > /tmp/open.json | |
| python3 - <<'PY' | |
| import datetime, json, os, re, subprocess | |
| repo = os.environ["REPO"] | |
| run_url = os.environ["RUN_URL"] | |
| grace = datetime.timedelta(minutes=int(os.environ["GRACE_MINUTES"])) | |
| components = ("web", "api", "docs", "gateway") | |
| # Upptime's incident body: "In [`<sha>`](<url>), <name> (<url>) was **down**:" | |
| # or "... experienced **degraded performance**:" followed by | |
| # "- HTTP code:" / "- Response time:" lines. | |
| checker_body = re.compile(r"^In \[`[0-9a-f]+`\]\(.*?\), .+ (was \*\*down\*\*|experienced \*\*degraded performance\*\*):", re.S) | |
| def parse_ts(value): | |
| return datetime.datetime.fromisoformat(value.replace("Z", "+00:00")) | |
| def history(slug): | |
| path = f"history/{slug}.yml" | |
| if not os.path.exists(path): | |
| return None | |
| fields = {} | |
| for line in open(path, encoding="utf-8"): | |
| if ":" in line and not line.startswith(" "): | |
| key, _, value = line.partition(":") | |
| fields[key.strip()] = value.strip() | |
| return fields | |
| issues = json.load(open("/tmp/open.json")) | |
| if not issues: | |
| print("No open status issues.") | |
| raise SystemExit(0) | |
| closed = 0 | |
| for issue in issues: | |
| number = issue["number"] | |
| labels = {label["name"] for label in issue["labels"]} | |
| slugs = [slug for slug in components if slug in labels] | |
| tag = f"#{number} {issue['title']!r}" | |
| if len(slugs) != 1: | |
| print(f"{tag}: skipped, component labels {sorted(slugs)} (need exactly one)") | |
| continue | |
| if not checker_body.match(issue.get("body") or ""): | |
| print(f"{tag}: skipped, not a checker-authored issue (manual incident)") | |
| continue | |
| slug = slugs[0] | |
| fields = history(slug) | |
| if not fields or "status" not in fields or "lastUpdated" not in fields: | |
| print(f"{tag}: skipped, history/{slug}.yml missing or unreadable") | |
| continue | |
| if fields["status"] != "up": | |
| print(f"{tag}: {slug} still reads {fields['status']!r}; leaving open") | |
| continue | |
| created = parse_ts(issue["createdAt"]) | |
| last = parse_ts(fields["lastUpdated"]) | |
| if last < created + grace: | |
| print(f"{tag}: {slug} reads up but lastUpdated {fields['lastUpdated']} is within " | |
| f"{grace} of created_at {issue['createdAt']}; leaving open") | |
| continue | |
| comment = ( | |
| "Closed by the stale-incident reconciler: the checker has read `up` since " | |
| f"{fields['lastUpdated']}; the run that recovered this incident failed before " | |
| f"closing it (push race). Component `{slug}`, opened {issue['createdAt']}. " | |
| f"Run: {run_url}" | |
| ) | |
| subprocess.run( | |
| ["gh", "issue", "close", str(number), "--repo", repo, "--comment", comment], | |
| check=True, | |
| ) | |
| print(f"{tag}: CLOSED ({slug} up since {fields['lastUpdated']})") | |
| closed += 1 | |
| print(f"Done: {closed} of {len(issues)} open status issue(s) closed.") | |
| PY |