Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,7 @@ stricter subset of Keep a Changelog).
- `--offline` flag for all subcommands: disables git fetch
- `oss-crs build-target` now validates `--bug-candidate`/`--bug-candidate-dir` the same way `oss-crs run` does: a missing path, or a directory passed to `--bug-candidate` (or a file passed to `--bug-candidate-dir`), fails before any container starts instead of being silently ignored.
- `libCRS` `apply_patch_build`: builder-side errors returned before a `rebuild_id` is assigned are now written to `<response_dir>/stderr.log` instead of being dropped. The public signature (`apply_patch_build(patch_path, response_dir, ...)`) and the positional shell form (`apply-patch-build <patch> <response_dir>`) are unchanged; `apply_patch_test` and `run-pov` are likewise unchanged.
- Added per-CRS token tracking (`prompt_tokens`/`completion_tokens`) alongside dollar spend.

### Added
- `oss-crs list-harnesses --fuzz-proj-path <PATH_TO_PROJ>` — builds an OSS-Fuzz project through its default Docker compile path and lists its runnable fuzz harnesses, reusing cached build artifacts until the project inputs or repository HEAD change.
Expand Down
14 changes: 14 additions & 0 deletions docs/design/architecture.md
Original file line number Diff line number Diff line change
Expand Up @@ -212,6 +212,20 @@ LiteLLM integration modes:
- **External mode**: OSS-CRS injects externally provided `OSS_CRS_LLM_API_URL` / `OSS_CRS_LLM_API_KEY_FILE`, and does not start internal LiteLLM sidecars.
- **Disabled mode** (`llm_config: null`): OSS-CRS performs no LiteLLM validation or sidecar setup.

#### Spend and token reporting

`litellm-key-gen` polls LiteLLM every few seconds and writes a host-recoverable `litellm-spend-report.json` per run (`<run_dir>/litellm-spend-report.json`):

```json
{"totals": {"credits_used": 0.0, "prompt_tokens": 251234,
"completion_tokens": 70112},
"crs": {"<crs-name>": {<same three fields>}},
"updated_at": 1788559707}
```

- `credits_used` is determined by the number/kind of tokens and LiteLLM's model cost map. `prompt_tokens`/`completion_tokens` come straight from provider `usage` in the spend logs. The total number of tokens is always prompt + completion. Models missing from the cost map report `credits_used == 0` alongside real token counts.
- The report lags live traffic by at most one poll interval: the sidecar performs a final forced poll on shutdown.

### Pinned Infrastructure Images

The LiteLLM and PostgreSQL container images are pinned by digest in
Expand Down
128 changes: 117 additions & 11 deletions oss-crs-infra/litellm-key-gen/main.py
Original file line number Diff line number Diff line change
@@ -1,9 +1,12 @@
#!/usr/bin/env python3
# SPDX-License-Identifier: MIT
import hashlib
import json
import os
import signal
import time
from datetime import datetime, timedelta, timezone

import yaml
import requests

Expand Down Expand Up @@ -106,26 +109,120 @@ def get_key_spend(api_key: str) -> float:
return 0.0


def collect_spend_summary(key_requests: dict[str, dict]) -> dict:
"""Build spend summary by querying per-key spend from LiteLLM."""
crs_summary: dict[str, dict[str, float]] = {}
total = 0.0
def _today_str() -> str:
return datetime.now(timezone.utc).date().isoformat()


def _end_date_str() -> str:
"""Exclusive window end for spend-log queries.

LiteLLM treats ``end_date`` as exclusive: querying with
``start_date == end_date == today`` always returns zero rows, so the
end of the window must be tomorrow to include today's traffic.
"""
return (datetime.now(timezone.utc).date() + timedelta(days=1)).isoformat()


def get_key_tokens(
api_key: str,
start_date: str | None = None,
end_date: str | None = None,
) -> tuple[int, int]:
"""Fetch cumulative (prompt, completion) tokens for a key.

Sums ``prompt_tokens``/``completion_tokens`` over this key's rows in
LiteLLM's spend logs (provider ``usage`` verbatim), independent of the
price map — so models missing from the cost map still report tokens
while ``spend`` stays 0. Rows are matched by key hash, the form
LiteLLM stores in ``api_key``. Fail-open: any error yields (0, 0) so
dollar polling never breaks.
"""
if not api_key:
return (0, 0)
headers = {
"Authorization": f"Bearer {LITELLM_MASTER_KEY}",
}
start = start_date or _today_str()
end = end_date or _end_date_str()
try:
key_hash = hashlib.sha256(api_key.encode()).hexdigest()
response = requests.get(
f"{LITELLM_API_URL}/spend/logs",
headers=headers,
params={"start_date": start, "end_date": end, "summarize": "false"},
timeout=30,
)
response.raise_for_status()
prompt = 0
completion = 0
for row in response.json():
if row.get("api_key") not in (api_key, key_hash):
continue
prompt += int(row.get("prompt_tokens", 0) or 0)
completion += int(row.get("completion_tokens", 0) or 0)
return prompt, completion
except Exception as e:
print(f"Error fetching token usage for key: {e}")
return (0, 0)


def collect_spend_summary(
key_requests: dict[str, dict],
start_date: str | None = None,
end_date: str | None = None,
) -> dict:
"""Build spend summary by querying per-key spend and tokens from LiteLLM."""
crs_summary: dict[str, dict[str, float | int]] = {}
total_spend = 0.0
total_prompt = 0
total_completion = 0
start = start_date or _today_str()
end = end_date or _end_date_str()
for crs_name, info in key_requests.items():
api_key = str(info.get("api_key", ""))
spend = get_key_spend(api_key) if api_key else 0.0
crs_summary[crs_name] = {"credits_used": round(spend, 6)}
total += spend
prompt, completion = get_key_tokens(api_key, start, end) if api_key else (0, 0)
crs_summary[crs_name] = {
"credits_used": round(spend, 6),
"prompt_tokens": prompt,
"completion_tokens": completion,
}
total_spend += spend
total_prompt += prompt
total_completion += completion
return {
"totals": {"credits_used": round(total, 6)},
"totals": {
"credits_used": round(total_spend, 6),
"prompt_tokens": total_prompt,
"completion_tokens": total_completion,
},
"crs": crs_summary,
"updated_at": int(time.time()),
}


def write_spend_summary(summary: dict) -> None:
with open(SPEND_REPORT_PATH, "w") as f:
"""Write the spend summary atomically.

A SIGKILL landing mid-write must never leave a truncated report behind
(readers turn that into zeros), so the payload goes to a temp file in
the same directory first and is moved into place with os.replace.
"""
tmp_path = SPEND_REPORT_PATH + ".tmp"
with open(tmp_path, "w") as f:
json.dump(summary, f, indent=2, sort_keys=True)
f.write("\n")
os.replace(tmp_path, SPEND_REPORT_PATH)


def _poll_and_persist(key_requests: dict[str, dict], start_date: str) -> None:
"""One poll cycle: collect the spend summary and persist it.

Shared by the polling loop and the shutdown path so the final flush
exercises exactly the same code as every poll.
"""
summary = collect_spend_summary(key_requests, start_date=start_date)
write_spend_summary(summary)


def main():
Expand Down Expand Up @@ -162,12 +259,21 @@ def main():
with open(READY_FILE_PATH, "w") as f:
f.write("ready\n")

# Poll LiteLLM spend and keep writing a host-recoverable summary file.
# Fixed window start so token queries capture the entire run.
# End date is refreshed every poll inside collect_spend_summary.
run_start_date = _today_str()

# Poll LiteLLM spend and tokens, keep writing host-recoverable summary.
while not _SHUTDOWN:
summary = collect_spend_summary(key_requests)
write_spend_summary(summary)
_poll_and_persist(key_requests, run_start_date)
time.sleep(max(SPEND_POLL_INTERVAL_SEC, 1))

# Final flush: capture traffic from the last partial interval
try:
_poll_and_persist(key_requests, run_start_date)
except Exception as e:
print(f"Error in final spend flush: {e}")

return 0


Expand Down
55 changes: 39 additions & 16 deletions oss-crs-infra/webui-publisher/main.py
Original file line number Diff line number Diff line change
Expand Up @@ -46,9 +46,7 @@
)
COVERAGE_BUILD_DIR = Path("/coverage_build")
LOG_DIR = Path("/webui_logs")
# LiteLLM spend report (mounted read-only when an LLM proxy is in the run).
# Written periodically by the litellm-key-gen sidecar; absent for LLM-free runs.
SPEND_REPORT_PATH = Path("/litellm-spend-report.json")
SPEND_REPORT_PATH = Path("/spend/litellm-spend-report.json")

POLL_INTERVAL = 5 # seconds
COVERAGE_INTERVAL = 30 # seconds
Expand Down Expand Up @@ -94,27 +92,52 @@ def _scan_dir(root: Path) -> dict[str, int]:
return counts


def read_cost() -> dict | None:
"""Read LLM spend from the litellm-key-gen spend report, if present.
def read_cost(report_path: Path | None = None) -> dict | None:
"""Read LLM spend and token counts from the litellm-key-gen report.

Returns ``{"total": float, "per_crs": {crs_name: float}}`` or ``None`` when
no spend report exists yet (e.g. LLM-free runs or before the first write).
Returns ``{"total": float, "per_crs": {...}, "prompt_tokens": int,
"completion_tokens": int,
"per_crs_tokens": {crs_name: {...}}}`` or ``None`` when no spend report
exists yet (e.g. LLM-free runs or before the first write). Token fields
default to 0 for old reports that predate token tracking; totals are
prompt + completion summed by the reader.
"""
if not SPEND_REPORT_PATH.is_file():
path = report_path if report_path is not None else SPEND_REPORT_PATH
if not path.is_file():
return None
try:
data = json.loads(SPEND_REPORT_PATH.read_text())
data = json.loads(path.read_text())
except (OSError, json.JSONDecodeError):
return None
total = data.get("totals", {}).get("credits_used")
per_crs = {
name: entry.get("credits_used", 0.0)
for name, entry in data.get("crs", {}).items()
if isinstance(entry, dict)
}
totals_raw = data.get("totals", {})
totals = totals_raw if isinstance(totals_raw, dict) else {}
crs_raw = data.get("crs", {})
crs = crs_raw if isinstance(crs_raw, dict) else {}
total = totals.get("credits_used")
entries = {name: entry for name, entry in crs.items() if isinstance(entry, dict)}
per_crs = {name: entry.get("credits_used", 0.0) for name, entry in entries.items()}
if total is None and not per_crs:
return None
return {"total": total, "per_crs": per_crs}
prompt = sum(entry.get("prompt_tokens", 0) for entry in entries.values())
completion = sum(entry.get("completion_tokens", 0) for entry in entries.values())
# Prefer the report's totals when present (old reports lack token fields).
totals_tokens = {
"prompt_tokens": totals.get("prompt_tokens", 0) or prompt,
"completion_tokens": totals.get("completion_tokens", 0) or completion,
}
per_crs_tokens = {
name: {
"prompt_tokens": entry.get("prompt_tokens", 0),
"completion_tokens": entry.get("completion_tokens", 0),
}
for name, entry in entries.items()
}
return {
"total": total,
"per_crs": per_crs,
**totals_tokens,
"per_crs_tokens": per_crs_tokens,
}


def build_snapshot() -> dict:
Expand Down
2 changes: 2 additions & 0 deletions oss_crs/src/config/artifacts.py
Original file line number Diff line number Diff line change
Expand Up @@ -75,6 +75,8 @@ class MetaArtifactCounts(BaseModel):

class MetaLLMStats(BaseModel):
credits_used: float = 0.0
prompt_tokens: int = 0
completion_tokens: int = 0


class MetaSidecarStats(BaseModel):
Expand Down
37 changes: 28 additions & 9 deletions oss_crs/src/crs_compose.py
Original file line number Diff line number Diff line change
Expand Up @@ -2156,11 +2156,19 @@ def _read_litellm_spend_summary(self, run_id: str, sanitizer: str) -> dict:
crs_raw = raw.get("crs")
crs = crs_raw if isinstance(crs_raw, dict) else {}
return {
"totals": {"credits_used": float(totals.get("credits_used", 0.0) or 0.0)},
"totals": {
"credits_used": totals.get("credits_used", 0.0),
"prompt_tokens": totals.get("prompt_tokens", 0),
"completion_tokens": totals.get("completion_tokens", 0),
},
"crs": {
name: {"credits_used": float((entry or {}).get("credits_used", 0.0))}
name: {
"credits_used": entry.get("credits_used", 0.0),
"prompt_tokens": entry.get("prompt_tokens", 0),
"completion_tokens": entry.get("completion_tokens", 0),
}
for name, entry in crs.items()
if isinstance(name, str)
if isinstance(name, str) and isinstance(entry, dict)
},
}

Expand Down Expand Up @@ -2207,7 +2215,11 @@ def _collect_run_meta(self, target: Target, run_id: str, sanitizer: str) -> dict
"artifacts": {
name.replace("-", "_"): 0 for name in SUBMITTED_ARTIFACT_DIR_NAMES
},
"llm": {"credits_used": 0.0},
"llm": {
"credits_used": 0.0,
"prompt_tokens": 0,
"completion_tokens": 0,
},
"sidecar": {
"patch_builds": 0,
"patch_tests": 0,
Expand All @@ -2223,12 +2235,11 @@ def _collect_run_meta(self, target: Target, run_id: str, sanitizer: str) -> dict
sidecar = self._read_sidecar_counts_for_crs(
crs.name, target, run_id, sanitizer
)
crs_llm = llm_summary.get("crs", {}).get(crs.name, {})
llm = {
"credits_used": float(
llm_summary.get("crs", {})
.get(crs.name, {})
.get("credits_used", 0.0)
)
"credits_used": crs_llm.get("credits_used", 0.0),
"prompt_tokens": crs_llm.get("prompt_tokens", 0),
"completion_tokens": crs_llm.get("completion_tokens", 0),
}

crs_meta[crs.name] = {
Expand All @@ -2241,13 +2252,21 @@ def _collect_run_meta(self, target: Target, run_id: str, sanitizer: str) -> dict
totals["artifacts"][key] += artifacts.get(key, 0)
for key in totals["sidecar"]:
totals["sidecar"][key] += sidecar.get(key, 0)
for key in ("prompt_tokens", "completion_tokens"):
totals["llm"][key] += llm.get(key, 0)

llm_total = float(llm_summary.get("totals", {}).get("credits_used", 0.0))
if llm_total == 0.0:
llm_total = round(
sum(crs_meta[name]["llm"]["credits_used"] for name in crs_meta), 6
)
totals["llm"]["credits_used"] = llm_total
# Prefer the report's totals when present (old reports lack tokens and
# read as 0); fall back to the per-CRS sum.
file_tokens = llm_summary.get("totals", {})
for key in ("prompt_tokens", "completion_tokens"):
if file_tokens.get(key):
totals["llm"][key] = file_tokens[key]

return {
"totals": totals,
Expand Down
4 changes: 3 additions & 1 deletion oss_crs/src/templates/renderer.py
Original file line number Diff line number Diff line change
Expand Up @@ -367,6 +367,8 @@ def render_run_crs_compose_docker_compose(
tmp_dir = tmp_docker_compose.dir if tmp_docker_compose.dir else Path("/tmp")
litellm_spend_report_path = str(tmp_dir / "litellm-spend-report.json")

litellm_spend_report_dir = str(Path(litellm_spend_report_path).parent)

context = {
"libCRS_path": str(LIBCRS_PATH),
"crs_compose_name": crs_compose_name,
Expand Down Expand Up @@ -413,7 +415,7 @@ def render_run_crs_compose_docker_compose(
"litellm_image": OSS_CRS_LITELLM_TAG,
"litellm_internal_url": LITELLM_INTERNAL_URL,
"offline": crs_compose.offline,
"litellm_spend_report_path": litellm_spend_report_path,
"litellm_spend_report_dir": litellm_spend_report_dir,
"postgres_image": OSS_CRS_POSTGRES_TAG,
"postgres_user": POSTGRES_USER,
"postgres_port": POSTGRES_PORT,
Expand Down
Loading