Skip to content

traffic-health

traffic-health #4461

# Hand-maintained (NOT generated by Upptime; update-template leaves it alone).
#
# Closes checker incidents that the checker already recovered but never closed.
#
# Why this exists: upptime/uptime-monitor commits history/<slug>.yml and then
# closes the incident issue in the SAME run. Its push does not rebase, so when
# another workflow's commit landed on main first the run died with
# "! [rejected] main -> main (fetch first)" AFTER the history file already
# said `status: up` locally; the next run re-pushed the history but, seeing
# the component up, skipped the incident step, and the issue stayed open until
# the NEXT transition. Issue #13 (2026-09-12) stayed open 17 h 26 min for a
# ~30-minute degradation that way, and Upptime counts open-to-close as
# downtime, so the API's uptime figures for two days were wrong.
#
# The push race itself is fixed by putting every workflow that pushes to main
# in Upptime's concurrency group (traffic-health.yml, server-latency.yml).
# This job is the safety net: for every OPEN `status`-labelled issue opened by
# the checker, read the component's history file from main; if it reads `up`
# and its lastUpdated is more than 10 minutes after the issue was opened, the
# recovery run failed before closing the issue, so close it here.
#
# Scope: checker-authored issues only (body matches Upptime's template).
# Manual incidents (a human opens an issue with the `status` label, README
# "Manual incident via issue") and `traffic-alert` issues (owned by
# traffic-health.yml) are never touched. Closes with github.token, so the
# comment is authored by github-actions[bot], not by the GH_PAT owner.
name: Stale Incidents
on:
# Rides the traffic-health dispatch the Vercel cron fires every 5 minutes
# (two minutes after the checker), so it runs right after a checker run
# would have closed the issue. GitHub's schedule is a fallback only.
repository_dispatch:
types: [traffic-health]
schedule:
- cron: "*/15 * * * *"
workflow_dispatch:
# Upptime's group: serializes this against the checker so the two never act
# on the same incident at once. Never cancel-in-progress in this group.
# Own group: this workflow only closes issues and never pushes to main, so it
# must not compete with the checker's push group (a pending run there is
# cancelled by the next arrival; two runs were cancelled that way on 2026-09-13).
concurrency:
group: ${{ github.repository }}-stale-incidents
cancel-in-progress: false
permissions:
contents: read
issues: write
env:
# Minutes the checker must have read `up` past the issue's created_at before
# the issue is treated as stranded rather than freshly opened.
GRACE_MINUTES: "10"
jobs:
reconcile:
name: Close incidents the checker recovered but left open
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- name: Checkout
uses: actions/checkout@v6
with:
# Always read the latest history from main, whatever ref fired us.
ref: main
- name: Reconcile open status issues against history/
env:
GH_TOKEN: ${{ github.token }}
REPO: ${{ github.repository }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
run: |
set -euo pipefail
gh issue list --repo "$REPO" --label status --state open --limit 100 \
--json number,title,createdAt,labels,body,url > /tmp/open.json
python3 - <<'PY'
import datetime, json, os, re, subprocess
repo = os.environ["REPO"]
run_url = os.environ["RUN_URL"]
grace = datetime.timedelta(minutes=int(os.environ["GRACE_MINUTES"]))
components = ("web", "api", "docs", "gateway")
# Upptime's incident body: "In [`<sha>`](<url>), <name> (<url>) was **down**:"
# or "... experienced **degraded performance**:" followed by
# "- HTTP code:" / "- Response time:" lines.
checker_body = re.compile(r"^In \[`[0-9a-f]+`\]\(.*?\), .+ (was \*\*down\*\*|experienced \*\*degraded performance\*\*):", re.S)
def parse_ts(value):
return datetime.datetime.fromisoformat(value.replace("Z", "+00:00"))
def history(slug):
path = f"history/{slug}.yml"
if not os.path.exists(path):
return None
fields = {}
for line in open(path, encoding="utf-8"):
if ":" in line and not line.startswith(" "):
key, _, value = line.partition(":")
fields[key.strip()] = value.strip()
return fields
issues = json.load(open("/tmp/open.json"))
if not issues:
print("No open status issues.")
raise SystemExit(0)
closed = 0
for issue in issues:
number = issue["number"]
labels = {label["name"] for label in issue["labels"]}
slugs = [slug for slug in components if slug in labels]
tag = f"#{number} {issue['title']!r}"
if len(slugs) != 1:
print(f"{tag}: skipped, component labels {sorted(slugs)} (need exactly one)")
continue
if not checker_body.match(issue.get("body") or ""):
print(f"{tag}: skipped, not a checker-authored issue (manual incident)")
continue
slug = slugs[0]
fields = history(slug)
if not fields or "status" not in fields or "lastUpdated" not in fields:
print(f"{tag}: skipped, history/{slug}.yml missing or unreadable")
continue
if fields["status"] != "up":
print(f"{tag}: {slug} still reads {fields['status']!r}; leaving open")
continue
created = parse_ts(issue["createdAt"])
last = parse_ts(fields["lastUpdated"])
if last < created + grace:
print(f"{tag}: {slug} reads up but lastUpdated {fields['lastUpdated']} is within "
f"{grace} of created_at {issue['createdAt']}; leaving open")
continue
comment = (
"Closed by the stale-incident reconciler: the checker has read `up` since "
f"{fields['lastUpdated']}; the run that recovered this incident failed before "
f"closing it (push race). Component `{slug}`, opened {issue['createdAt']}. "
f"Run: {run_url}"
)
subprocess.run(
["gh", "issue", "close", str(number), "--repo", repo, "--comment", comment],
check=True,
)
print(f"{tag}: CLOSED ({slug} up since {fields['lastUpdated']})")
closed += 1
print(f"Done: {closed} of {len(issues)} open status issue(s) closed.")
PY