Compare commits

...
Author SHA1 Message Date
esmeetu 4b6ff605fc update frontend
Signed-off-by: esmeetu <jasonailu87@gmail.com>
2026-03-08 04:23:28 +00:00
esmeetu c5eacd261d Merge remote-tracking branch 'origin/main' into vllm-dashboard 2026-03-07 15:08:32 +00:00
esmeetu 549b8242d6 refactor
Signed-off-by: esmeetu <jasonailu87@gmail.com>
2026-03-03 09:21:45 +00:00
esmeetu 3fde6cc382 Merge branch 'main' into vllm-dashboard
Signed-off-by: esmeetu <jasonailu87@gmail.com>
2026-03-03 08:31:16 +00:00
esmeetu f0b536640c Merge branch 'main' into vllm-dashboard
Signed-off-by: esmeetu <jasonailu87@gmail.com>
2026-03-01 12:28:39 +08:00
Roy WangandGitHub 9e173a5cdc Merge branch 'main' into vllm-dashboard 2026-02-03 13:16:14 +08:00
Roy WangandGitHub 09a977e897 Merge branch 'main' into vllm-dashboard 2026-01-29 16:51:43 +08:00
esmeetu 1b2b0043d8 add more metrics
Signed-off-by: esmeetu <jasonailu87@gmail.com>
2026-01-15 21:15:37 +08:00
esmeetu a87a2676f2 add multi-modal support test
Signed-off-by: esmeetu <jasonailu87@gmail.com>
2026-01-15 19:34:19 +08:00
esmeetu f54775736a add params
Signed-off-by: esmeetu <jasonailu87@gmail.com>
2026-01-15 19:33:22 +08:00
esmeetu 934e6b2a46 update style & add filter
Signed-off-by: esmeetu <jasonailu87@gmail.com>
2026-01-15 19:09:47 +08:00
esmeetu e9966f1c52 remove engine stats
Signed-off-by: esmeetu <jasonailu87@gmail.com>
2026-01-15 18:53:46 +08:00
esmeetu 12e25a1d3f fix null stats
Signed-off-by: esmeetu <jasonailu87@gmail.com>
2026-01-15 18:41:41 +08:00
esmeetu 3bebb3d705 refactor webui & add Engine Stats
Signed-off-by: esmeetu <jasonailu87@gmail.com>
2026-01-15 18:34:22 +08:00
esmeetu 36547719b0 support healthcheck
Signed-off-by: esmeetu <jasonailu87@gmail.com>
2026-01-15 17:56:36 +08:00
esmeetu d4d3e90c3e Add --enable-dashboard for vLLM Web UI
Signed-off-by: esmeetu <jasonailu87@gmail.com>
2026-01-13 18:49:10 +08:00
7 changed files with 3817 additions and 6 deletions
+292
View File
@@ -0,0 +1,292 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
"""Tests for vLLM dashboard functionality."""
import os
from unittest.mock import AsyncMock, MagicMock
import pytest
from fastapi import FastAPI
from fastapi.testclient import TestClient
from vllm.entrypoints.serve.dashboard.api_router import (
_get_dashboard_html,
_get_explicit_cli_args,
_get_vllm_env_vars,
attach_router,
router,
)
from vllm.version import __version__ as VLLM_VERSION
class TestDashboardHelpers:
"""Test helper functions in dashboard api_router."""
def test_get_vllm_env_vars(self):
"""Test that _get_vllm_env_vars returns VLLM environment variables."""
vllm_envs, explicit_envs = _get_vllm_env_vars()
# Should return a dict of VLLM_* variables
assert isinstance(vllm_envs, dict)
assert isinstance(explicit_envs, set)
# All keys should start with VLLM_
for key in vllm_envs:
assert key.startswith("VLLM_"), f"Key {key} doesn't start with VLLM_"
# Should not include keys with "KEY" (secrets)
for key in vllm_envs:
assert "KEY" not in key, f"Key {key} contains KEY (potential secret)"
def test_get_vllm_env_vars_explicit_detection(self):
"""Test that explicitly set env vars are detected."""
test_var = "VLLM_TEST_DASHBOARD_VAR"
try:
os.environ[test_var] = "test_value"
# Need to reload envs module to pick up the new var
# For this test, we just verify the mechanism works
_, explicit_envs = _get_vllm_env_vars()
# The test var won't be in vllm.envs, but the mechanism is tested
assert isinstance(explicit_envs, set)
finally:
os.environ.pop(test_var, None)
def test_get_explicit_cli_args_none(self):
"""Test _get_explicit_cli_args with None args."""
result = _get_explicit_cli_args(None)
assert result == set()
def test_get_explicit_cli_args_with_args(self):
"""Test _get_explicit_cli_args with mock args."""
mock_args = MagicMock()
mock_args.model = "test-model"
mock_args.served_model_name = "test-served-model"
mock_args.dtype = "float16" # Non-default value
result = _get_explicit_cli_args(mock_args)
# model should always be explicit if set
assert "model" in result
assert "served_model_name" in result
def test_get_dashboard_html(self):
"""Test that dashboard HTML is loaded correctly."""
html = _get_dashboard_html()
assert isinstance(html, str)
assert len(html) > 0
assert "<!DOCTYPE html>" in html
assert "vLLM Dashboard" in html
assert "</html>" in html
class TestDashboardRouter:
"""Test dashboard router endpoints."""
@pytest.fixture
def app(self):
"""Create a test FastAPI app with dashboard router."""
app = FastAPI()
app.include_router(router)
# Mock app state
app.state.openai_serving_models = None
app.state.vllm_config = None
app.state.args = None
app.state.engine_client = None
return app
@pytest.fixture
def client(self, app):
"""Create a test client."""
return TestClient(app)
def test_dashboard_index(self, client):
"""Test GET /dashboard returns HTML page."""
response = client.get("/dashboard")
assert response.status_code == 200
assert "text/html" in response.headers["content-type"]
assert "vLLM Dashboard" in response.text
def test_dashboard_api_info(self, client):
"""Test GET /dashboard/api/info returns server info."""
response = client.get("/dashboard/api/info")
assert response.status_code == 200
data = response.json()
assert "version" in data
assert data["version"] == VLLM_VERSION
assert "status" in data
assert data["status"] == "running"
def test_dashboard_api_info_with_models(self, app):
"""Test /dashboard/api/info includes model info when available."""
# Mock serving models
mock_model = MagicMock()
mock_model.id = "test-model"
mock_model.root = "test-model-root"
mock_models_response = MagicMock()
mock_models_response.data = [mock_model]
mock_serving_models = AsyncMock()
mock_serving_models.show_available_models = AsyncMock(
return_value=mock_models_response
)
app.state.openai_serving_models = mock_serving_models
client = TestClient(app)
response = client.get("/dashboard/api/info")
assert response.status_code == 200
data = response.json()
assert "models" in data
assert len(data["models"]) == 1
assert data["models"][0]["id"] == "test-model"
assert data["models"][0]["root"] == "test-model-root"
def test_dashboard_api_info_health_check(self, app):
"""Test /dashboard/api/info uses engine health check."""
# Mock healthy engine client
mock_engine_client = AsyncMock()
mock_engine_client.check_health = AsyncMock(return_value=None)
mock_engine_client.is_sleeping = AsyncMock(return_value=False)
mock_engine_client.is_paused = AsyncMock(return_value=False)
app.state.engine_client = mock_engine_client
client = TestClient(app)
response = client.get("/dashboard/api/info")
assert response.status_code == 200
data = response.json()
assert data["status"] == "running"
assert data["is_sleeping"] is False
assert data["is_paused"] is False
def test_dashboard_api_info_unhealthy_engine(self, app):
"""Test /dashboard/api/info reports unhealthy when engine is dead."""
from vllm.v1.engine.exceptions import EngineDeadError
# Mock unhealthy engine client
mock_engine_client = AsyncMock()
mock_engine_client.check_health = AsyncMock(side_effect=EngineDeadError())
mock_engine_client.is_sleeping = AsyncMock(return_value=False)
mock_engine_client.is_paused = AsyncMock(return_value=False)
app.state.engine_client = mock_engine_client
client = TestClient(app)
response = client.get("/dashboard/api/info")
assert response.status_code == 200
data = response.json()
assert data["status"] == "unhealthy"
def test_dashboard_api_info_sleeping_engine(self, app):
"""Test /dashboard/api/info reports is_sleeping when engine is sleeping."""
# Mock sleeping engine client
mock_engine_client = AsyncMock()
mock_engine_client.check_health = AsyncMock(return_value=None)
mock_engine_client.is_sleeping = AsyncMock(return_value=True)
mock_engine_client.is_paused = AsyncMock(return_value=False)
app.state.engine_client = mock_engine_client
client = TestClient(app)
response = client.get("/dashboard/api/info")
assert response.status_code == 200
data = response.json()
assert data["status"] == "running"
assert data["is_sleeping"] is True
def test_dashboard_api_info_server_load(self, app):
"""Test /dashboard/api/info includes server_load when available."""
app.state.server_load_metrics = 5
client = TestClient(app)
response = client.get("/dashboard/api/info")
assert response.status_code == 200
data = response.json()
assert data["server_load"] == 5
def test_dashboard_api_info_paused_engine(self, app):
"""Test /dashboard/api/info reports is_paused when engine is paused."""
# Mock paused engine client
mock_engine_client = AsyncMock()
mock_engine_client.check_health = AsyncMock(return_value=None)
mock_engine_client.is_sleeping = AsyncMock(return_value=False)
mock_engine_client.is_paused = AsyncMock(return_value=True)
app.state.engine_client = mock_engine_client
client = TestClient(app)
response = client.get("/dashboard/api/info")
assert response.status_code == 200
data = response.json()
assert data["status"] == "running"
assert data["is_paused"] is True
def test_dashboard_api_metrics(self, client):
"""Test GET /dashboard/api/metrics returns metrics."""
response = client.get("/dashboard/api/metrics")
assert response.status_code == 200
data = response.json()
# Should return a dict (may be empty if no metrics available)
assert isinstance(data, dict)
class TestAttachRouter:
"""Test attach_router function."""
def test_attach_router_disabled(self):
"""Test router is not attached when dashboard is disabled."""
app = FastAPI()
mock_args = MagicMock()
mock_args.enable_dashboard = False
app.state.args = mock_args
attach_router(app)
# Dashboard routes should not be available
client = TestClient(app)
response = client.get("/dashboard")
assert response.status_code == 404
def test_attach_router_enabled(self):
"""Test router is attached when dashboard is enabled."""
app = FastAPI()
mock_args = MagicMock()
mock_args.enable_dashboard = True
app.state.args = mock_args
app.state.openai_serving_models = None
app.state.vllm_config = None
attach_router(app)
# Dashboard routes should be available
client = TestClient(app)
response = client.get("/dashboard")
assert response.status_code == 200
def test_attach_router_no_args(self):
"""Test router is not attached when args is None."""
app = FastAPI()
app.state.args = None
attach_router(app)
# Dashboard routes should not be available
client = TestClient(app)
response = client.get("/dashboard")
assert response.status_code == 404
+5
View File
@@ -281,6 +281,11 @@ class FrontendArgs(BaseFrontendArgs):
Enable offline FastAPI documentation for air-gapped environments.
Uses vendored static assets bundled with vLLM.
"""
enable_dashboard: bool = False
"""
Enable the vLLM web dashboard at /dashboard for monitoring and testing.
Provides server info, metrics display, and a chat interface for testing.
"""
use_gpu_for_pooling_score: bool = False
"""If set, run pooling score MaxSim on GPU in the API server process.
Can significantly improve late-interaction scoring performance.
+13 -6
View File
@@ -129,8 +129,8 @@ def get_uvicorn_log_config(args: Namespace) -> dict | None:
Priority:
1. If log_config_file is specified, use it
2. If disable_access_log_for_endpoints is specified, create a config with
the access log filter
2. If disable_access_log_for_endpoints is specified or dashboard is enabled,
create a config with the access log filter
3. Otherwise, return None (use uvicorn defaults)
"""
# First, try to load from file if specified
@@ -138,16 +138,23 @@ def get_uvicorn_log_config(args: Namespace) -> dict | None:
if log_config is not None:
return log_config
# If endpoints to filter are specified, create a config with the filter
if args.disable_access_log_for_endpoints:
from vllm.logging_utils import create_uvicorn_log_config
excluded_paths: list[str] = []
# Parse comma-separated string into list
# User-specified endpoints to filter
if args.disable_access_log_for_endpoints:
excluded_paths = [
p.strip()
for p in args.disable_access_log_for_endpoints.split(",")
if p.strip()
]
# Dashboard polling endpoints create noise every 5s; suppress them
if getattr(args, "enable_dashboard", False):
excluded_paths.extend(["/dashboard/api/info", "/dashboard/api/metrics", "/dashboard"])
if excluded_paths:
from vllm.logging_utils import create_uvicorn_log_config
return create_uvicorn_log_config(
excluded_paths=excluded_paths,
log_level=args.uvicorn_log_level,
+6
View File
@@ -52,6 +52,12 @@ def register_vllm_serve_api_routers(app: FastAPI):
attach_tokenize_router(app)
from vllm.entrypoints.serve.dashboard.api_router import (
attach_router as attach_dashboard_router,
)
attach_dashboard_router(app)
from .instrumentator import register_instrumentator_api_routers
register_instrumentator_api_routers(app)
@@ -0,0 +1,2 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
@@ -0,0 +1,458 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
"""
Dashboard router for vLLM web UI.
Provides endpoints for:
- /dashboard - Main dashboard HTML page
- /dashboard/api/info - Server information JSON (config, env, load, status)
- /dashboard/api/metrics - Metrics JSON
"""
import asyncio
import functools
import pathlib
import pydantic
from fastapi import APIRouter, FastAPI, Request
from fastapi.responses import HTMLResponse, JSONResponse
import vllm.envs as envs
from vllm.config import VllmConfig
from vllm.logger import init_logger
from vllm.version import __version__ as VLLM_VERSION
logger = init_logger(__name__)
router = APIRouter()
PydanticVllmConfig = pydantic.TypeAdapter(VllmConfig)
def _get_vllm_env_vars() -> tuple[dict, set]:
"""Get all VLLM environment variables and which ones are explicitly set."""
import os
from vllm.config.utils import normalize_value
vllm_envs = {}
explicit_envs = set()
for key in dir(envs):
if key.startswith("VLLM_") and "KEY" not in key:
value = getattr(envs, key, None)
if value is not None:
value = normalize_value(value)
vllm_envs[key] = value
# Check if explicitly set in environment
if key in os.environ:
explicit_envs.add(key)
return vllm_envs, explicit_envs
def _get_explicit_cli_args(args) -> tuple[set, dict]:
"""Get CLI args that were explicitly set by user (non-default values).
Returns a tuple of (set of explicit arg names, dict of arg name -> value).
"""
from vllm.config.utils import normalize_value
from vllm.engine.arg_utils import AsyncEngineArgs
from vllm.entrypoints.openai.cli_args import FrontendArgs
explicit_args: set[str] = set()
non_default_args: dict = {}
if args is None:
return explicit_args, non_default_args
# Get default values from dataclasses
try:
frontend_defaults = FrontendArgs()
engine_defaults = AsyncEngineArgs(model="")
for key, default_val in vars(frontend_defaults).items():
if hasattr(args, key):
current_val = getattr(args, key)
if current_val != default_val:
explicit_args.add(key)
non_default_args[key] = normalize_value(current_val)
for key, default_val in vars(engine_defaults).items():
if hasattr(args, key) and key != "model":
current_val = getattr(args, key)
if current_val != default_val:
explicit_args.add(key)
non_default_args[key] = normalize_value(current_val)
# model is always explicit if set
if hasattr(args, "model") and args.model:
explicit_args.add("model")
non_default_args["model"] = args.model
if hasattr(args, "served_model_name") and args.served_model_name:
explicit_args.add("served_model_name")
non_default_args["served_model_name"] = normalize_value(
args.served_model_name
)
except Exception as e:
logger.debug("Failed to get explicit CLI args: %s", e)
return explicit_args, non_default_args
_resolved_attn_backend: str | None = None
def _resolve_attn_backend_name(vllm_config, configured_backend) -> str | None:
"""Best-effort resolution of the actual attention backend.
When the user leaves the backend on auto (None), we call the platform's
selection logic with the model's head_size / dtype so the dashboard can
show the concrete backend (e.g. FLASH_ATTN) instead of "auto".
Returns the resolved backend name string or None on failure.
"""
global _resolved_attn_backend
if _resolved_attn_backend is not None:
return _resolved_attn_backend
try:
from vllm.platforms import current_platform
from vllm.v1.attention.backends.registry import AttentionBackendEnum
from vllm.v1.attention.selector import AttentionSelectorConfig
mc = vllm_config.model_config
cc = vllm_config.cache_config
selector_cfg = AttentionSelectorConfig(
head_size=mc.get_head_size(),
dtype=mc.dtype,
kv_cache_dtype=cc.cache_dtype if cc.cache_dtype != "auto" else None,
block_size=getattr(cc, "block_size", None),
use_mla=mc.use_mla,
)
cls_path = current_platform.get_attn_backend_cls(
configured_backend,
attn_selector_config=selector_cfg,
)
if not cls_path:
return None
for member in AttentionBackendEnum:
if member.value and member.value == cls_path:
_resolved_attn_backend = member.name
return _resolved_attn_backend
_resolved_attn_backend = cls_path.rsplit(".", 1)[-1].replace(
"Backend", ""
)
return _resolved_attn_backend
except Exception as e:
logger.debug("Failed to resolve attention backend: %s", e)
return None
def _get_startup_info(vllm_config) -> dict:
"""Derive startup info from vllm_config available in the API server.
Exposes values that are computed properties or enum-typed in the config
and would otherwise be missing or opaque in the JSON-serialized output.
Also computes KV cache capacity and max concurrency from cache_config
(num_gpu_blocks is set via the engine READY handshake).
"""
info: dict = {}
if vllm_config is None:
return info
# Architecture is a @property, not serialized by Pydantic
try:
mc = vllm_config.model_config
info["architecture"] = mc.architecture
except Exception:
pass
# Attention backend — resolve the actual backend even when config is auto
try:
ac = vllm_config.attention_config
configured = getattr(ac, "backend", None)
resolved_name = _resolve_attn_backend_name(vllm_config, configured)
if resolved_name:
info["attention_backend"] = resolved_name
elif configured is not None:
info["attention_backend"] = configured.name
except Exception:
pass
# Optimization level
try:
opt = vllm_config.optimization_level
if opt is not None:
info["optimization_level"] = f"O{int(opt)}"
except Exception:
pass
# Compilation config: mode, cudagraph, inductor partition, fusion passes
try:
cpc = vllm_config.compilation_config
mode = getattr(cpc, "mode", None)
if mode is not None:
info["compilation_mode"] = mode.name
cgm = getattr(cpc, "cudagraph_mode", None)
if cgm is not None:
info["cudagraph_mode"] = cgm.name
igp = getattr(cpc, "use_inductor_graph_partition", None)
if igp is not None:
info["inductor_graph_partition"] = igp
pc = getattr(cpc, "pass_config", None)
if pc is not None:
fusions = {}
for attr in (
"fuse_norm_quant",
"fuse_act_quant",
"fuse_attn_quant",
"enable_sp",
"fuse_gemm_comms",
"fuse_allreduce_rms",
"enable_qk_norm_rope_fusion",
):
val = getattr(pc, attr, None)
if val is True:
fusions[attr] = True
if fusions:
info["pass_config_fusions"] = fusions
except Exception:
pass
# Kernel config: flashinfer autotune, MoE backend
try:
kc = vllm_config.kernel_config
fi = getattr(kc, "enable_flashinfer_autotune", None)
if fi is not None:
info["flashinfer_autotune"] = fi
moe = getattr(kc, "moe_backend", None)
if moe is not None:
info["moe_backend"] = moe
except Exception:
pass
# KV cache capacity and max concurrency
try:
cc = vllm_config.cache_config
mc = vllm_config.model_config
num_gpu_blocks = getattr(cc, "num_gpu_blocks", None)
block_size = getattr(cc, "block_size", None)
if num_gpu_blocks and block_size:
kv_cache_tokens = num_gpu_blocks * block_size
info["kv_cache_tokens"] = kv_cache_tokens
max_model_len = getattr(mc, "max_model_len", None)
if max_model_len and max_model_len > 0:
info["max_concurrency"] = round(
kv_cache_tokens / max_model_len, 2
)
except Exception:
pass
return info
@functools.lru_cache(maxsize=1)
def _get_system_env_info_cached() -> dict:
"""Get system environment info (GPU, CUDA, torch, OS, etc.).
Cached since this info never changes during the lifetime of the server.
"""
from vllm.collect_env import get_env_info
return get_env_info()._asdict()
def _get_dashboard_html() -> str:
"""Load the dashboard HTML from static file."""
static_dir = pathlib.Path(__file__).parent / "static"
html_file = static_dir / "index.html"
if html_file.exists():
return html_file.read_text(encoding="utf-8")
return "<html><body><h1>Dashboard HTML not found</h1></body></html>"
@router.get("/dashboard", response_class=HTMLResponse)
async def dashboard_index() -> HTMLResponse:
"""Serve the main dashboard page."""
html_content = _get_dashboard_html()
return HTMLResponse(content=html_content)
@router.get("/dashboard/api/info")
async def dashboard_info(request: Request) -> JSONResponse:
"""Get server information for dashboard display."""
from vllm.v1.engine.exceptions import EngineDeadError
info: dict = {
"version": VLLM_VERSION,
"status": "running",
}
# Check engine health
try:
engine_client = getattr(request.app.state, "engine_client", None)
if engine_client is not None:
await engine_client.check_health()
info["status"] = "running"
except EngineDeadError:
info["status"] = "unhealthy"
except Exception as e:
logger.debug("Failed to check engine health: %s", e)
# Get model information
try:
serving_models = request.app.state.openai_serving_models
if serving_models is not None:
models_response = await serving_models.show_available_models()
info["models"] = [
{"id": model.id, "root": model.root} for model in models_response.data
]
except Exception as e:
logger.warning("Failed to get model info for dashboard: %s", e)
info["models"] = []
# Get full engine config and derive startup info
vllm_config = getattr(request.app.state, "vllm_config", None)
try:
if vllm_config is not None:
info["vllm_config"] = PydanticVllmConfig.dump_python(
vllm_config, mode="json", fallback=str
)
except Exception as e:
logger.warning("Failed to get vllm_config for dashboard: %s", e)
try:
startup_info = _get_startup_info(vllm_config)
if startup_info:
info["startup_info"] = startup_info
except Exception as e:
logger.debug("Failed to get startup_info for dashboard: %s", e)
# Get environment variables (with explicit markers)
try:
vllm_env, explicit_envs = _get_vllm_env_vars()
info["vllm_env"] = vllm_env
info["explicit_envs"] = list(explicit_envs)
except Exception as e:
logger.warning("Failed to get vllm_env for dashboard: %s", e)
# Get system environment info (GPU, CUDA, torch, OS, etc.)
try:
info["system_env"] = await asyncio.to_thread(
_get_system_env_info_cached
)
except Exception as e:
logger.debug("Failed to get system_env for dashboard: %s", e)
# Get explicit CLI args and non-default args dict
try:
args = getattr(request.app.state, "args", None)
explicit_args, non_default_args = _get_explicit_cli_args(args)
info["explicit_args"] = list(explicit_args)
info["non_default_args"] = non_default_args
except Exception as e:
logger.warning("Failed to get explicit_args for dashboard: %s", e)
# Get server load metrics
try:
server_load = getattr(request.app.state, "server_load_metrics", None)
if server_load is not None:
info["server_load"] = server_load
except Exception as e:
logger.debug("Failed to get server_load for dashboard: %s", e)
# Check if engine is sleeping (only available in dev mode)
try:
engine_client = getattr(request.app.state, "engine_client", None)
if engine_client is not None:
is_sleeping_method = getattr(engine_client, "is_sleeping", None)
if is_sleeping_method is not None:
info["is_sleeping"] = await is_sleeping_method()
except Exception as e:
logger.debug("Failed to check is_sleeping for dashboard: %s", e)
# Check if engine is paused (for RLHF workflows)
try:
engine_client = getattr(request.app.state, "engine_client", None)
if engine_client is not None:
is_paused_method = getattr(engine_client, "is_paused", None)
if is_paused_method is not None:
info["is_paused"] = await is_paused_method()
except Exception as e:
logger.debug("Failed to check is_paused for dashboard: %s", e)
return JSONResponse(content=info)
@router.get("/dashboard/api/metrics")
async def dashboard_metrics(request: Request) -> JSONResponse:
"""Get metrics for dashboard display.
Returns both Prometheus metrics and internal engine stats that are only
accessible in-process (not available via external /metrics endpoint).
"""
metrics: dict = {}
# Try to get metrics from prometheus registry
try:
from vllm.v1.metrics.reader import (
Counter,
Gauge,
Histogram,
get_metrics_snapshot,
)
snapshot = get_metrics_snapshot()
for metric in snapshot:
name = metric.name
# For labeled metrics, create unique keys to avoid overwriting
# e.g. vllm:request_success with finished_reason label
if metric.labels:
# Filter out common labels (model_name, engine) for key
key_labels = {
k: v
for k, v in metric.labels.items()
if k not in ("model_name", "engine")
}
if key_labels:
# Create key like "vllm:request_success:stop"
label_suffix = ":".join(str(v) for v in key_labels.values())
name = f"{metric.name}:{label_suffix}"
if isinstance(metric, (Counter, Gauge)):
metrics[name] = {
"value": metric.value,
"labels": metric.labels,
}
elif isinstance(metric, Histogram):
metrics[name] = {
"count": metric.count,
"sum": metric.sum,
"labels": metric.labels,
}
except ImportError:
logger.debug("Metrics reader not available")
except Exception as e:
logger.warning("Failed to get metrics for dashboard: %s", e)
# Get server load if tracking is enabled
try:
server_load = getattr(request.app.state, "server_load_metrics", None)
if server_load is not None:
metrics["server_load"] = {"value": server_load, "labels": {}}
except Exception:
pass
return JSONResponse(content=metrics)
def attach_router(app: FastAPI) -> None:
"""Attach dashboard router if enabled via args."""
args = getattr(app.state, "args", None)
if args is None or not getattr(args, "enable_dashboard", False):
return
logger.info("Enabling vLLM dashboard at /dashboard")
app.include_router(router)
File diff suppressed because it is too large Load Diff