Skip to content

traffic-health

traffic-health #6974

# Hand-maintained (NOT generated by Upptime; update-template leaves it alone).
#
# Health from REAL customer traffic, not synthetic pings. Every 5 minutes this
# reads the gateway ledger (gateway_requests) over the last 15 minutes and
# asks two questions the checker cannot:
#
# 1. Is the gateway itself failing requests? "Gateway-owned" failures are the
# terminal classes the platform is responsible for (internal, unavailable),
# never the customer's (quota_exceeded, invalid_request) or the upstream
# provider's (provider_internal, throttled, timeout).
# 2. Did traffic collapse? Request volume is compared with the median volume
# for the same 15-minute slot at this hour over the prior 7 days.
#
# The verdict and the figures land in assets/status-ui/traffic-health.json,
# which the status-ui overlay renders on the API row (fetched from raw main, so
# no redeploy is needed). A non-ok verdict opens ONE GitHub issue labeled
# `traffic-alert` (assigned to the owner) and posts to Slack when
# NOTIFICATION_SLACK_WEBHOOK_URL is configured; recovery closes the issue and
# posts again. Upptime's own `status`-labeled incidents are untouched, so the
# uptime percentages and 90-day bars keep their single source.
#
# The read uses the hard-capped read-only ops_agent role (connection limit 8,
# role-level statement timeout); one aggregate query per run. If the database
# itself is unreachable twice in a row that is treated as DOWN: the gateway's
# readiness gates on a live database ping, so an unreachable database is an API
# outage, not a monitoring glitch.
name: Traffic Health
on:
# The schedule is a fallback: GitHub runs it only when it has capacity (the
# checker's identical cron landed about every two hours). The Vercel cron
# (vercel/api/cron/dispatch.js) fires the repository_dispatch every 5 minutes.
schedule:
- cron: "*/5 * * * *"
repository_dispatch:
types: [traffic-health]
workflow_dispatch:
# Same group as Upptime's generated workflows (uptime, response-time, graphs,
# summary). Anything that pushes to main MUST share it: upptime/uptime-monitor
# pushes without rebasing, so a traffic-health push landing while the checker
# ran killed the checker's run ("! [rejected] main -> main (fetch first)"),
# and when that happened on a recovery the incident issue stayed open for
# hours (issue #13, 2026-09-12). Serializing the pushes removes the race; the
# rebase loop in "Commit if changed" stays as a second line of defence.
concurrency:
group: ${{ github.repository }}-${{ github.head_ref || github.ref_name }}-upptime
cancel-in-progress: false
permissions:
contents: write
issues: write
jobs:
refresh:
name: Read live gateway traffic
runs-on: ubuntu-latest
timeout-minutes: 8
steps:
- name: Checkout
uses: actions/checkout@v6
with:
token: ${{ secrets.GH_PAT || github.token }}
- name: Query the gateway ledger
id: query
env:
DB_URL: ${{ secrets.PROD_OPS_AGENT_DB_URL }}
run: |
set -uo pipefail
row=$(PGCONNECT_TIMEOUT=10 PGOPTIONS="-c statement_timeout=20s" psql "$DB_URL" -X -qtA -v ON_ERROR_STOP=1 -c "
with cur as (
select count(*) as total,
count(*) filter (where terminal_state is not null) as terminal,
count(*) filter (where terminal_state = 'completed') as completed,
count(*) filter (where terminal_failure_class in ('internal', 'unavailable')) as gateway_owned
from gateway_requests
where accepted_at >= now() - interval '15 minutes'
), base as (
select coalesce(percentile_cont(0.5) within group (order by n), 0) as median_15m
from (
select date_trunc('hour', accepted_at)
+ (floor(extract(minute from accepted_at) / 15) * 15) * interval '1 minute' as bucket,
count(*) as n
from gateway_requests
where accepted_at >= now() - interval '7 days'
and accepted_at < now() - interval '15 minutes'
and extract(hour from accepted_at) = extract(hour from now())
group by 1
) s
)
select cur.total, cur.terminal, cur.completed, cur.gateway_owned, round(base.median_15m)
from cur, base" 2>/tmp/psql.err | tr -d '\n')
status=${PIPESTATUS[0]}
if [ "$status" -ne 0 ]; then
echo "database read failed:" >&2
sed -E 's#(postgres(ql)?://)[^@]*@#\1***@#' /tmp/psql.err >&2 || true
echo "db_ok=false" >> "$GITHUB_OUTPUT"
exit 0
fi
echo "db_ok=true" >> "$GITHUB_OUTPUT"
echo "row=$row" >> "$GITHUB_OUTPUT"
- name: Compute the verdict and refresh the JSON
id: verdict
env:
DB_OK: ${{ steps.query.outputs.db_ok }}
ROW: ${{ steps.query.outputs.row }}
run: |
python3 - <<'EOF'
import datetime, json, os
path = "assets/status-ui/traffic-health.json"
previous = json.load(open(path)) if os.path.exists(path) else {}
now = datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
data = {
"generatedAt": now,
"windowMinutes": 15,
"notes": (
"Health derived from real customer traffic in the gateway ledger "
"(gateway_requests), refreshed every 5 minutes by the traffic-health "
"workflow. gatewayErrorPct counts only failures the gateway owns "
"(internal, unavailable); customer rejections (quota, invalid request) "
"and upstream provider errors are excluded. baselineRequests is the "
"7-day median request count for this 15-minute slot at this hour."
),
}
db_failures = int(previous.get("consecutiveDbFailures", 0))
if os.environ["DB_OK"] != "true":
db_failures += 1
if previous.get("figures"):
data["figures"] = previous["figures"]
data["consecutiveDbFailures"] = db_failures
data["verdict"] = "down" if db_failures >= 2 else previous.get("verdict", "ok")
data["reason"] = (
"The production database has been unreachable on consecutive checks; "
"the gateway cannot serve without it."
if db_failures >= 2
else "One database read failed; holding the previous verdict until the next check."
)
else:
total, terminal, completed, gateway_owned, baseline = (
int(float(x)) for x in os.environ["ROW"].strip().split("|")
)
err_pct = (100.0 * gateway_owned / terminal) if terminal else 0.0
data["consecutiveDbFailures"] = 0
data["figures"] = {
"requests": total,
"terminal": terminal,
"completed": completed,
"gatewayOwnedFailures": gateway_owned,
"gatewayErrorPct": round(err_pct, 2),
"baselineRequests": baseline,
}
# Thresholds (owner-tunable): the gateway failing a quarter of a
# meaningful sample, or serving nothing when it normally serves
# plenty, is DOWN; a twentieth, or a fifth of normal volume, is
# DEGRADED. Small samples never alert.
if (terminal >= 20 and err_pct >= 25.0) or (total == 0 and baseline >= 20):
data["verdict"] = "down"
elif (terminal >= 20 and err_pct >= 5.0) or (baseline >= 50 and total < 0.2 * baseline):
data["verdict"] = "degraded"
else:
data["verdict"] = "ok"
if data["verdict"] == "ok":
data["reason"] = ""
elif total == 0 or (baseline >= 50 and total < 0.2 * baseline):
data["reason"] = f"Traffic collapsed: {total} requests in 15 minutes against a usual {baseline}."
else:
data["reason"] = (
f"The gateway failed {gateway_owned} of {terminal} finished requests "
f"({err_pct:.1f}%) in the last 15 minutes."
)
open(path, "w").write(json.dumps(data, indent=2) + "\n")
with open(os.environ["GITHUB_OUTPUT"], "a") as out:
out.write(f"verdict={data['verdict']}\n")
out.write(f"reason={data.get('reason', '')}\n")
out.write(f"previous={previous.get('verdict', 'ok')}\n")
EOF
- name: Commit if changed
run: |
git config user.name "Upptime Bot"
git config user.email "73812536+upptime-bot@users.noreply.github.com"
git add assets/status-ui/traffic-health.json
git diff --cached --quiet && { echo "No change."; exit 0; }
git commit -m ":stethoscope: Refresh live traffic health [skip ci]"
# Upptime's own bots commit to main every few minutes; rebase over
# whatever landed while this job ran instead of failing the push.
for _ in 1 2 3; do
git push && exit 0
git pull --rebase --autostash && continue
done
exit 1
- name: Open, hold, or close the traffic alert
env:
GH_TOKEN: ${{ secrets.GH_PAT || github.token }}
SLACK_WEBHOOK: ${{ secrets.NOTIFICATION_SLACK_WEBHOOK_URL }}
VERDICT: ${{ steps.verdict.outputs.verdict }}
REASON: ${{ steps.verdict.outputs.reason }}
REPO: ${{ github.repository }}
RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}
run: |
set -euo pipefail
open_issue=$(gh issue list --repo "$REPO" --label traffic-alert --state open --json number --jq '.[0].number // empty')
notify() {
[ -n "${SLACK_WEBHOOK:-}" ] || { echo "No Slack webhook configured; skipping Slack."; return 0; }
payload=$(python3 -c 'import json,sys; print(json.dumps({"text": sys.argv[1]}))' "$1")
curl -sS -X POST -H 'Content-type: application/json' --data "$payload" "$SLACK_WEBHOOK" >/dev/null || echo "Slack post failed." >&2
}
case "$VERDICT" in
down|degraded)
label=$([ "$VERDICT" = down ] && echo "API DOWN" || echo "API degraded")
if [ -z "$open_issue" ]; then
gh label create traffic-alert --repo "$REPO" --color d93f0b \
--description "Opened by the traffic-health workflow from live gateway traffic" 2>/dev/null || true
url=$(gh issue create --repo "$REPO" --label traffic-alert --assignee SilenNaihin \
--title "$label: live gateway traffic ($VERDICT)" \
--body "$(printf '%s\n\nSource: real customer traffic in the gateway ledger, last 15 minutes.\nVerdict: **%s**\n\nThis issue closes automatically when the next check reads healthy. Run: %s' "$REASON" "$VERDICT" "$RUN_URL")")
notify ":rotating_light: *$label* (status.experientiallabs.ai, live traffic): $REASON $url"
else
gh issue comment "$open_issue" --repo "$REPO" --body "Still **$VERDICT**: $REASON ($RUN_URL)"
fi
;;
*)
if [ -n "$open_issue" ]; then
gh issue close "$open_issue" --repo "$REPO" --comment "Recovered: live gateway traffic reads healthy again ($RUN_URL)."
notify ":white_check_mark: *API recovered* (status.experientiallabs.ai, live traffic): gateway-owned errors back under threshold."
fi
;;
esac