traffic-health #6974
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # Hand-maintained (NOT generated by Upptime; update-template leaves it alone). | |
| # | |
| # Health from REAL customer traffic, not synthetic pings. Every 5 minutes this | |
| # reads the gateway ledger (gateway_requests) over the last 15 minutes and | |
| # asks two questions the checker cannot: | |
| # | |
| # 1. Is the gateway itself failing requests? "Gateway-owned" failures are the | |
| # terminal classes the platform is responsible for (internal, unavailable), | |
| # never the customer's (quota_exceeded, invalid_request) or the upstream | |
| # provider's (provider_internal, throttled, timeout). | |
| # 2. Did traffic collapse? Request volume is compared with the median volume | |
| # for the same 15-minute slot at this hour over the prior 7 days. | |
| # | |
| # The verdict and the figures land in assets/status-ui/traffic-health.json, | |
| # which the status-ui overlay renders on the API row (fetched from raw main, so | |
| # no redeploy is needed). A non-ok verdict opens ONE GitHub issue labeled | |
| # `traffic-alert` (assigned to the owner) and posts to Slack when | |
| # NOTIFICATION_SLACK_WEBHOOK_URL is configured; recovery closes the issue and | |
| # posts again. Upptime's own `status`-labeled incidents are untouched, so the | |
| # uptime percentages and 90-day bars keep their single source. | |
| # | |
| # The read uses the hard-capped read-only ops_agent role (connection limit 8, | |
| # role-level statement timeout); one aggregate query per run. If the database | |
| # itself is unreachable twice in a row that is treated as DOWN: the gateway's | |
| # readiness gates on a live database ping, so an unreachable database is an API | |
| # outage, not a monitoring glitch. | |
| name: Traffic Health | |
| on: | |
| # The schedule is a fallback: GitHub runs it only when it has capacity (the | |
| # checker's identical cron landed about every two hours). The Vercel cron | |
| # (vercel/api/cron/dispatch.js) fires the repository_dispatch every 5 minutes. | |
| schedule: | |
| - cron: "*/5 * * * *" | |
| repository_dispatch: | |
| types: [traffic-health] | |
| workflow_dispatch: | |
| # Same group as Upptime's generated workflows (uptime, response-time, graphs, | |
| # summary). Anything that pushes to main MUST share it: upptime/uptime-monitor | |
| # pushes without rebasing, so a traffic-health push landing while the checker | |
| # ran killed the checker's run ("! [rejected] main -> main (fetch first)"), | |
| # and when that happened on a recovery the incident issue stayed open for | |
| # hours (issue #13, 2026-09-12). Serializing the pushes removes the race; the | |
| # rebase loop in "Commit if changed" stays as a second line of defence. | |
| concurrency: | |
| group: ${{ github.repository }}-${{ github.head_ref || github.ref_name }}-upptime | |
| cancel-in-progress: false | |
| permissions: | |
| contents: write | |
| issues: write | |
| jobs: | |
| refresh: | |
| name: Read live gateway traffic | |
| runs-on: ubuntu-latest | |
| timeout-minutes: 8 | |
| steps: | |
| - name: Checkout | |
| uses: actions/checkout@v6 | |
| with: | |
| token: ${{ secrets.GH_PAT || github.token }} | |
| - name: Query the gateway ledger | |
| id: query | |
| env: | |
| DB_URL: ${{ secrets.PROD_OPS_AGENT_DB_URL }} | |
| run: | | |
| set -uo pipefail | |
| row=$(PGCONNECT_TIMEOUT=10 PGOPTIONS="-c statement_timeout=20s" psql "$DB_URL" -X -qtA -v ON_ERROR_STOP=1 -c " | |
| with cur as ( | |
| select count(*) as total, | |
| count(*) filter (where terminal_state is not null) as terminal, | |
| count(*) filter (where terminal_state = 'completed') as completed, | |
| count(*) filter (where terminal_failure_class in ('internal', 'unavailable')) as gateway_owned | |
| from gateway_requests | |
| where accepted_at >= now() - interval '15 minutes' | |
| ), base as ( | |
| select coalesce(percentile_cont(0.5) within group (order by n), 0) as median_15m | |
| from ( | |
| select date_trunc('hour', accepted_at) | |
| + (floor(extract(minute from accepted_at) / 15) * 15) * interval '1 minute' as bucket, | |
| count(*) as n | |
| from gateway_requests | |
| where accepted_at >= now() - interval '7 days' | |
| and accepted_at < now() - interval '15 minutes' | |
| and extract(hour from accepted_at) = extract(hour from now()) | |
| group by 1 | |
| ) s | |
| ) | |
| select cur.total, cur.terminal, cur.completed, cur.gateway_owned, round(base.median_15m) | |
| from cur, base" 2>/tmp/psql.err | tr -d '\n') | |
| status=${PIPESTATUS[0]} | |
| if [ "$status" -ne 0 ]; then | |
| echo "database read failed:" >&2 | |
| sed -E 's#(postgres(ql)?://)[^@]*@#\1***@#' /tmp/psql.err >&2 || true | |
| echo "db_ok=false" >> "$GITHUB_OUTPUT" | |
| exit 0 | |
| fi | |
| echo "db_ok=true" >> "$GITHUB_OUTPUT" | |
| echo "row=$row" >> "$GITHUB_OUTPUT" | |
| - name: Compute the verdict and refresh the JSON | |
| id: verdict | |
| env: | |
| DB_OK: ${{ steps.query.outputs.db_ok }} | |
| ROW: ${{ steps.query.outputs.row }} | |
| run: | | |
| python3 - <<'EOF' | |
| import datetime, json, os | |
| path = "assets/status-ui/traffic-health.json" | |
| previous = json.load(open(path)) if os.path.exists(path) else {} | |
| now = datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") | |
| data = { | |
| "generatedAt": now, | |
| "windowMinutes": 15, | |
| "notes": ( | |
| "Health derived from real customer traffic in the gateway ledger " | |
| "(gateway_requests), refreshed every 5 minutes by the traffic-health " | |
| "workflow. gatewayErrorPct counts only failures the gateway owns " | |
| "(internal, unavailable); customer rejections (quota, invalid request) " | |
| "and upstream provider errors are excluded. baselineRequests is the " | |
| "7-day median request count for this 15-minute slot at this hour." | |
| ), | |
| } | |
| db_failures = int(previous.get("consecutiveDbFailures", 0)) | |
| if os.environ["DB_OK"] != "true": | |
| db_failures += 1 | |
| if previous.get("figures"): | |
| data["figures"] = previous["figures"] | |
| data["consecutiveDbFailures"] = db_failures | |
| data["verdict"] = "down" if db_failures >= 2 else previous.get("verdict", "ok") | |
| data["reason"] = ( | |
| "The production database has been unreachable on consecutive checks; " | |
| "the gateway cannot serve without it." | |
| if db_failures >= 2 | |
| else "One database read failed; holding the previous verdict until the next check." | |
| ) | |
| else: | |
| total, terminal, completed, gateway_owned, baseline = ( | |
| int(float(x)) for x in os.environ["ROW"].strip().split("|") | |
| ) | |
| err_pct = (100.0 * gateway_owned / terminal) if terminal else 0.0 | |
| data["consecutiveDbFailures"] = 0 | |
| data["figures"] = { | |
| "requests": total, | |
| "terminal": terminal, | |
| "completed": completed, | |
| "gatewayOwnedFailures": gateway_owned, | |
| "gatewayErrorPct": round(err_pct, 2), | |
| "baselineRequests": baseline, | |
| } | |
| # Thresholds (owner-tunable): the gateway failing a quarter of a | |
| # meaningful sample, or serving nothing when it normally serves | |
| # plenty, is DOWN; a twentieth, or a fifth of normal volume, is | |
| # DEGRADED. Small samples never alert. | |
| if (terminal >= 20 and err_pct >= 25.0) or (total == 0 and baseline >= 20): | |
| data["verdict"] = "down" | |
| elif (terminal >= 20 and err_pct >= 5.0) or (baseline >= 50 and total < 0.2 * baseline): | |
| data["verdict"] = "degraded" | |
| else: | |
| data["verdict"] = "ok" | |
| if data["verdict"] == "ok": | |
| data["reason"] = "" | |
| elif total == 0 or (baseline >= 50 and total < 0.2 * baseline): | |
| data["reason"] = f"Traffic collapsed: {total} requests in 15 minutes against a usual {baseline}." | |
| else: | |
| data["reason"] = ( | |
| f"The gateway failed {gateway_owned} of {terminal} finished requests " | |
| f"({err_pct:.1f}%) in the last 15 minutes." | |
| ) | |
| open(path, "w").write(json.dumps(data, indent=2) + "\n") | |
| with open(os.environ["GITHUB_OUTPUT"], "a") as out: | |
| out.write(f"verdict={data['verdict']}\n") | |
| out.write(f"reason={data.get('reason', '')}\n") | |
| out.write(f"previous={previous.get('verdict', 'ok')}\n") | |
| EOF | |
| - name: Commit if changed | |
| run: | | |
| git config user.name "Upptime Bot" | |
| git config user.email "73812536+upptime-bot@users.noreply.github.com" | |
| git add assets/status-ui/traffic-health.json | |
| git diff --cached --quiet && { echo "No change."; exit 0; } | |
| git commit -m ":stethoscope: Refresh live traffic health [skip ci]" | |
| # Upptime's own bots commit to main every few minutes; rebase over | |
| # whatever landed while this job ran instead of failing the push. | |
| for _ in 1 2 3; do | |
| git push && exit 0 | |
| git pull --rebase --autostash && continue | |
| done | |
| exit 1 | |
| - name: Open, hold, or close the traffic alert | |
| env: | |
| GH_TOKEN: ${{ secrets.GH_PAT || github.token }} | |
| SLACK_WEBHOOK: ${{ secrets.NOTIFICATION_SLACK_WEBHOOK_URL }} | |
| VERDICT: ${{ steps.verdict.outputs.verdict }} | |
| REASON: ${{ steps.verdict.outputs.reason }} | |
| REPO: ${{ github.repository }} | |
| RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} | |
| run: | | |
| set -euo pipefail | |
| open_issue=$(gh issue list --repo "$REPO" --label traffic-alert --state open --json number --jq '.[0].number // empty') | |
| notify() { | |
| [ -n "${SLACK_WEBHOOK:-}" ] || { echo "No Slack webhook configured; skipping Slack."; return 0; } | |
| payload=$(python3 -c 'import json,sys; print(json.dumps({"text": sys.argv[1]}))' "$1") | |
| curl -sS -X POST -H 'Content-type: application/json' --data "$payload" "$SLACK_WEBHOOK" >/dev/null || echo "Slack post failed." >&2 | |
| } | |
| case "$VERDICT" in | |
| down|degraded) | |
| label=$([ "$VERDICT" = down ] && echo "API DOWN" || echo "API degraded") | |
| if [ -z "$open_issue" ]; then | |
| gh label create traffic-alert --repo "$REPO" --color d93f0b \ | |
| --description "Opened by the traffic-health workflow from live gateway traffic" 2>/dev/null || true | |
| url=$(gh issue create --repo "$REPO" --label traffic-alert --assignee SilenNaihin \ | |
| --title "$label: live gateway traffic ($VERDICT)" \ | |
| --body "$(printf '%s\n\nSource: real customer traffic in the gateway ledger, last 15 minutes.\nVerdict: **%s**\n\nThis issue closes automatically when the next check reads healthy. Run: %s' "$REASON" "$VERDICT" "$RUN_URL")") | |
| notify ":rotating_light: *$label* (status.experientiallabs.ai, live traffic): $REASON $url" | |
| else | |
| gh issue comment "$open_issue" --repo "$REPO" --body "Still **$VERDICT**: $REASON ($RUN_URL)" | |
| fi | |
| ;; | |
| *) | |
| if [ -n "$open_issue" ]; then | |
| gh issue close "$open_issue" --repo "$REPO" --comment "Recovered: live gateway traffic reads healthy again ($RUN_URL)." | |
| notify ":white_check_mark: *API recovered* (status.experientiallabs.ai, live traffic): gateway-owned errors back under threshold." | |
| fi | |
| ;; | |
| esac |