Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 9 additions & 3 deletions docker/Dockerfile
Original file line number Diff line number Diff line change
Expand Up @@ -29,7 +29,7 @@ RUN cargo install qmassa --version 1.3.2 --locked
RUN apt-get clean && rm -rf /var/lib/apt/lists/*

# Install Python packages
COPY requirements.txt .
COPY docker/requirements.txt .
RUN pip install --require-hashes --no-deps --no-cache-dir --break-system-packages -r requirements.txt

# Clean up any old PCM
Expand All @@ -50,10 +50,16 @@ RUN echo "Installing PCM" && \


# Copy all scripts and supervisor config
COPY scripts/ /scripts/
COPY supervisord.conf supervisord.conf
COPY docker/scripts/ /scripts/
COPY docker/supervisord.conf supervisord.conf

# Ensure all .sh are executable
RUN chmod +x /scripts/*.sh

# live-metrics: opt-in live HTTP API served alongside the existing collectors.
COPY live-metrics/ /live-metrics/
# --ignore-installed: fastapi pulls in a newer typing-extensions
RUN pip install --no-cache-dir --break-system-packages --ignore-installed -r /live-metrics/requirements.txt
Comment thread
d-rushma marked this conversation as resolved.
EXPOSE 9000

CMD ["/usr/bin/supervisord", "-c", "/supervisord.conf"]
6 changes: 4 additions & 2 deletions docker/docker-compose.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -8,15 +8,17 @@ version: '3'
services:
benchmark:
build:
context: .
dockerfile: Dockerfile
context: ..
dockerfile: docker/Dockerfile
image: benchmark:latest
container_name: metrics-collector
privileged: true
pid: host
network_mode: "host"
environment:
- NPU_LOG=/tmp/results/npu_usage.csv
- RESULTS_DIR=/tmp/results
- METRICS_HTTP_PORT=9000
volumes:
- ${log_dir}:/tmp/results
- /tmp/.X11-unix:/tmp/.X11-unix
Expand Down
34 changes: 30 additions & 4 deletions docker/scripts/collect_gpu.sh
Original file line number Diff line number Diff line change
@@ -1,4 +1,15 @@
#!/bin/bash
#
# RESULTS_DIR defaults to /tmp/results, matching the previous hardcoded path
RESULTS_DIR="${RESULTS_DIR:-/tmp/results}"
mkdir -p "${RESULTS_DIR}"

# 0 (default) reproduces the original behavior exactly: qmassa runs once,
# unbounded, for the life of the script -- fine for a benchmark run that
# lasts minutes. Set > 0 to make qmassa exit and restart every N seconds,
# deleting its own output file first, so a 24/7 live dashboard doesn't
# eventually OOM on an ever-growing JSON file (qmassa never rotates it).
QMASSA_CYCLE_SECONDS="${QMASSA_CYCLE_SECONDS:-0}"

# Get all lines containing pci: and both device= and card=
mapfile -t pci_devices < <(
Expand Down Expand Up @@ -40,12 +51,27 @@ for device_line in "${pci_devices[@]}"; do
if [[ -n "$device_id" && -n "$card_num" ]]; then
echo "Valid device found: $pci_info | Device ID: $device_id | Card Number: $card_num"

output_file="/tmp/results/qmassa${card_num}-${device_id}-${driver}-tool-generated.json"
touch $output_file
chown 1000:1000 $output_file
output_file="${RESULTS_DIR}/qmassa${card_num}-${device_id}-${driver}-tool-generated.json"
touch "$output_file"
chown 1000:1000 "$output_file"

echo "Starting igt capture to $output_file"
$HOME/.cargo/bin/qmassa -d $pci_info -g -x -t "$output_file" 2>> /tmp/results/qmassa_error.log
if [ "$QMASSA_CYCLE_SECONDS" -gt 0 ] 2>/dev/null; then
# Bounded-cycle mode: run for QMASSA_CYCLE_SECONDS, then drop the
# file and start fresh, so a long-running live dashboard reader
# never sees an unbounded (multi-GB) JSON document.
while true; do
timeout "${QMASSA_CYCLE_SECONDS}s" "$HOME/.cargo/bin/qmassa" \
-d $pci_info -g -x -t "$output_file" 2>> "${RESULTS_DIR}/qmassa_error.log"
rm -f "$output_file"
touch "$output_file"
chown 1000:1000 "$output_file"
done
else
# Legacy/default behavior: single unbounded invocation, matching
# every existing Docker-mode benchmarking consumer exactly.
$HOME/.cargo/bin/qmassa -d $pci_info -g -x -t "$output_file" 2>> "${RESULTS_DIR}/qmassa_error.log"
fi
else
echo "Skipping $card: Incomplete pci info"
fi
Expand Down
41 changes: 23 additions & 18 deletions docker/scripts/collect_platform.sh
Original file line number Diff line number Diff line change
Expand Up @@ -4,43 +4,48 @@
#
# SPDX-License-Identifier: Apache-2.0
#
# RESULTS_DIR defaults to /tmp/results, matching the previous hardcoded
# path exactly -- callers that don't set it (e.g. docker-compose.yaml) see
# no behavior change.
RESULTS_DIR="${RESULTS_DIR:-/tmp/results}"
mkdir -p "${RESULTS_DIR}"

echo "Starting platform data collection"

echo "Starting sar collection"
touch /tmp/results/cpu_usage.log
chown 1000:1000 /tmp/results/cpu_usage.log
sar 1 >& /tmp/results/cpu_usage.log &
touch "${RESULTS_DIR}/cpu_usage.log"
chown 1000:1000 "${RESULTS_DIR}/cpu_usage.log"
sar 1 >& "${RESULTS_DIR}/cpu_usage.log" &

echo "Starting free collection"
touch /tmp/results/memory_usage.log
chown 1000:1000 /tmp/results/memory_usage.log
free -s 1 >& /tmp/results/memory_usage.log &
touch "${RESULTS_DIR}/memory_usage.log"
chown 1000:1000 "${RESULTS_DIR}/memory_usage.log"
free -s 1 >& "${RESULTS_DIR}/memory_usage.log" &

echo "Starting iotop collection"
touch /tmp/results/disk_bandwidth.log
chown 1000:1000 /tmp/results/disk_bandwidth.log
iotop -o -P -b >& /tmp/results/disk_bandwidth.log &
touch "${RESULTS_DIR}/disk_bandwidth.log"
chown 1000:1000 "${RESULTS_DIR}/disk_bandwidth.log"
iotop -o -P -b >& "${RESULTS_DIR}/disk_bandwidth.log" &

is_xeon=`lscpu | grep -i xeon | wc -l`

if [ "$is_xeon" == "1" ]
then
echo "Starting pcm-memory collection"
touch /tmp/results/pcm-memory.csv
chown 1000:1000 /tmp/results/pcm-memory.csv
/opt/intel/pcm-bin/bin/pcm-memory 1 -silent -nc -csv=/tmp/results/pcm-memory.csv &
touch "${RESULTS_DIR}/pcm-memory.csv"
chown 1000:1000 "${RESULTS_DIR}/pcm-memory.csv"
/opt/intel/pcm-bin/bin/pcm-memory 1 -silent -nc -csv="${RESULTS_DIR}/pcm-memory.csv" &

echo "Starting pcm-power collection"
touch /tmp/results/pcm-power.log
chown 1000:1000 /tmp/results/pcm-power.log
/opt/intel/pcm-bin/bin/pcm-power >& /tmp/results/pcm-power.log &
touch "${RESULTS_DIR}/pcm-power.log"
chown 1000:1000 "${RESULTS_DIR}/pcm-power.log"
/opt/intel/pcm-bin/bin/pcm-power >& "${RESULTS_DIR}/pcm-power.log" &
fi

echo "Starting general pcm collection"
touch /tmp/results/pcm.csv
chown 1000:1000 /tmp/results/pcm.csv
/opt/intel/pcm-bin/bin/pcm 1 -silent -r -nc -nsys -csv=/tmp/results/pcm.csv &
touch "${RESULTS_DIR}/pcm.csv"
chown 1000:1000 "${RESULTS_DIR}/pcm.csv"
/opt/intel/pcm-bin/bin/pcm 1 -silent -r -nc -nsys -csv="${RESULTS_DIR}/pcm.csv" &

while true
do
Expand Down
6 changes: 6 additions & 0 deletions docker/supervisord.conf
Original file line number Diff line number Diff line change
Expand Up @@ -19,3 +19,9 @@ command=/scripts/collect_platform.sh
autorestart=true
redirect_stderr=true
stdout_logfile=/dev/stdout

[program:metrics_api]
command=python3 /live-metrics/metrics_api.py
autorestart=true
redirect_stderr=true
stdout_logfile=/dev/stdout
53 changes: 53 additions & 0 deletions live-metrics/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,53 @@
# live-metrics

A small HTTP API that turns the flat files written by
`docker/scripts/collect_platform.sh`, `collect_gpu.sh` and `collect_npu.py`
into a live, pollable JSON time series (CPU / GPU / NPU / memory / power).

`performance-tools` already ran those collectors inside a privileged Docker
container (`docker/Dockerfile` + `docker/docker-compose.yaml`, used today by
`loss-prevention` as a submodule) for **offline benchmarking** --
`benchmark-scripts/consolidate_multiple_run_of_metrics.py` only parses the
results *after* a run finishes. `live-metrics/` adds a second way to consume
the same collectors: a **live HTTP API** (`metrics_api.py` /
`metrics_parser.py`), served from inside the same container via an
additional `supervisord` program, so a UI can poll `GET /metrics`
continuously while the collectors keep running -- not just after a
benchmark completes.

Both use cases run **the exact same collector scripts** under
`docker/scripts/` -- `live-metrics/` only adds a reader on top, it doesn't
change what's collected or how.

## Running it

```bash
cd docker
log_dir=./results docker compose build # picks up live-metrics/ automatically, see docker/Dockerfile
log_dir=./results docker compose up -d
curl http://localhost:9000/metrics | jq
```
`docker-compose.yaml` uses `network_mode: host`, so port 9000 is reachable
directly on the host once the container is up — no port publishing needed.

## Environment variables

| Variable | Default | Meaning |
|---|---|---|
| `RESULTS_DIR` | `/tmp/results` | Where the collectors write, and where `metrics_parser.py` reads from. |
| `NPU_LOG` | `${RESULTS_DIR}/npu_usage.csv` | Override if you need the NPU CSV somewhere else. |
| `METRICS_HTTP_PORT` | `9000` | Port `metrics_api.py` listens on. |
| `DEVICE_CONFIG_PATH` | unset | Optional path to a `KEY=VALUE` file for `GET /device-config`; returns `{}` if unset. |
| `QMASSA_CYCLE_SECONDS` | `0` (disabled) | If set > 0, `collect_gpu.sh` restarts qmassa every N seconds and deletes its output file first, keeping the JSON bounded for long-running live dashboards. `0` reproduces the original, unbounded, single-invocation behavior used by existing offline-benchmarking consumers. **Set this for any 24/7 live-dashboard deployment** (e.g. `180`) -- qmassa never rotates its own output file, and it will otherwise grow unbounded (observed ~600 MB/hour in one production deployment). |
| `METRICS_GPU_MAX_JSON_MB` | `256` | `build_gpu_series()` refuses to `json.load()` a qmassa file larger than this (defense in depth alongside `QMASSA_CYCLE_SECONDS`). |
| `METRICS_GPU_MAX_POINTS` | `300` | Caps how many qmassa samples are parsed per request (only the tail is ever plotted). |

## Endpoints

| Method & path | Returns |
|---|---|
| `GET /health` | `{"status": "ok"}` |
| `GET /metrics` | `{"cpu_utilization": [...], "gpu_utilization": [...], "npu_utilization": [...], "memory": [...], "power": [...]}` |
| `GET /platform-info` | `{"Processor": "...", "iGPU": "...", "NPU": "...", "Memory": "...", "Storage": "..."}` |
| `GET /memory` | Latest single memory snapshot |
| `GET /device-config` | `{}` unless `DEVICE_CONFIG_PATH` is set (see above) |
91 changes: 91 additions & 0 deletions live-metrics/metrics_api.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,91 @@
'''
* Copyright (C) 2026 Intel Corporation.
*
* SPDX-License-Identifier: Apache-2.0
'''

"""
Live metrics HTTP API for performance-tools.

Served inside the Docker image alongside the collectors (see
docker/supervisord.conf's [program:metrics_api]); this process only ever
reads $RESULTS_DIR, it has no other dependency on how the collectors were
started.

Endpoints
---------
GET /health -> 200 {"status": "ok"}
GET /metrics -> time-series JSON (cpu / gpu / npu / memory / power)
GET /platform-info -> hardware summary (processor, iGPU, NPU, memory, storage)
GET /memory -> latest single memory snapshot
GET /device-config -> optional per-workload device summary (see metrics_parser.build_device_config_payload)
"""

import os

from fastapi import FastAPI
from fastapi.middleware.cors import CORSMiddleware

from metrics_parser import (
build_device_config_payload,
build_metrics_payload,
get_platform_info,
parse_memory_usage,
)

app = FastAPI(title="performance-tools live-metrics")

app.add_middleware(
CORSMiddleware,
allow_origins=[
o.strip()
for o in os.getenv("METRICS_CORS_ORIGINS", "*").split(",")
if o.strip()
],
allow_credentials=False,
allow_methods=["*"],
allow_headers=["*"],
)


@app.get("/health")
def health() -> dict[str, str]:
"""Health check endpoint."""
return {"status": "ok"}


@app.get("/metrics")
def metrics() -> dict:
"""Time-series utilization data: CPU, GPU, NPU, memory, power."""
return build_metrics_payload()


@app.get("/platform-info")
def platform_info() -> dict:
"""Hardware summary: processor, iGPU, NPU, memory, storage."""
return get_platform_info()


@app.get("/memory")
def memory() -> dict:
"""Latest single memory snapshot."""
data = parse_memory_usage()
return data if data is not None else {"error": "no memory data"}


@app.get("/device-config")
def device_config() -> dict:
"""Optional per-workload device summary; {} if not configured."""
return build_device_config_payload()


if __name__ == "__main__":
import uvicorn

port = int(os.getenv("METRICS_HTTP_PORT", "9000"))
print(
f"[live-metrics] starting on 0.0.0.0:{port} "
f"endpoints: /health /metrics /platform-info /memory /device-config",
flush=True,
)
uvicorn.run("metrics_api:app", host="0.0.0.0", port=port, reload=False)
Loading
Loading