Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 0 additions & 24 deletions fastdeploy/entrypoints/openai/api_server.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,6 @@
"""

import asyncio
import json
import os
import signal
import threading
Expand Down Expand Up @@ -572,29 +571,6 @@ async def metrics():
return Response(metrics_text, media_type="text/plain")


@metrics_app.get("/config-info")
def config_info() -> Response:
"""
Get the current configuration of the API server.
"""
global llm_engine
if llm_engine is None:
return Response("Engine not loaded", status_code=500)
cfg = llm_engine.cfg

def process_object(obj):
if hasattr(obj, "__dict__"):
# 处理有__dict__属性的对象
return obj.__dict__
return None # 或其他默认处理

cfg_dict = {k: v for k, v in cfg.__dict__.items()}
env_dict = {k: v() for k, v in environment_variables.items()}
cfg_dict["env_config"] = env_dict
result_content = json.dumps(cfg_dict, default=process_object, ensure_ascii=False)
return Response(result_content, media_type="application/json")


def run_metrics_server():
"""
run metrics server
Expand Down
7 changes: 0 additions & 7 deletions tests/e2e/test_ernie_03b_pd_router_v0.py
Original file line number Diff line number Diff line change
Expand Up @@ -259,13 +259,6 @@ def headers():
return {"Content-Type": "application/json"}


def test_metrics_config(metrics_url):
timeout = 600
url = metrics_url.replace("metrics", "config-info")
res = requests.get(url, timeout=timeout)
assert res.status_code == 200


def send_request(url, payload, timeout=60):
"""
发送请求到指定的URL,并返回响应结果。
Expand Down
7 changes: 0 additions & 7 deletions tests/e2e/test_ernie_03b_pd_router_v1_ipc.py
Original file line number Diff line number Diff line change
Expand Up @@ -260,13 +260,6 @@ def headers():
return {"Content-Type": "application/json"}


def test_metrics_config(metrics_url):
timeout = 600
url = metrics_url.replace("metrics", "config-info")
res = requests.get(url, timeout=timeout)
assert res.status_code == 200


def send_request(url, payload, timeout=60):
"""
发送请求到指定的URL,并返回响应结果。
Expand Down
7 changes: 0 additions & 7 deletions tests/e2e/test_ernie_03b_pd_router_v1_rdma_tp1.py
Original file line number Diff line number Diff line change
Expand Up @@ -264,13 +264,6 @@ def headers():
return {"Content-Type": "application/json"}


def test_metrics_config(metrics_url):
timeout = 600
url = metrics_url.replace("metrics", "config-info")
res = requests.get(url, timeout=timeout)
assert res.status_code == 200


def send_request(url, payload, timeout=60):
"""
发送请求到指定的URL,并返回响应结果。
Expand Down
7 changes: 0 additions & 7 deletions tests/e2e/test_ernie_03b_pd_router_v1_rdma_tp2.py
Original file line number Diff line number Diff line change
Expand Up @@ -268,13 +268,6 @@ def headers():
return {"Content-Type": "application/json"}


def test_metrics_config(metrics_url):
timeout = 600
url = metrics_url.replace("metrics", "config-info")
res = requests.get(url, timeout=timeout)
assert res.status_code == 200


def send_request(url, payload, timeout=60):
"""
发送请求到指定的URL,并返回响应结果。
Expand Down
7 changes: 0 additions & 7 deletions tests/e2e/test_ernie_03b_router.py
Original file line number Diff line number Diff line change
Expand Up @@ -265,13 +265,6 @@ def headers():
return {"Content-Type": "application/json"}


def test_metrics_config(metrics_url):
timeout = 600
url = metrics_url.replace("metrics", "config-info")
res = requests.get(url, timeout=timeout)
assert res.status_code == 200


def send_request(url, payload, timeout=60):
"""
发送请求到指定的URL,并返回响应结果。
Expand Down
129 changes: 1 addition & 128 deletions tests/entrypoints/openai/test_metrics_routes.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,6 @@

import asyncio
import importlib
import json
import os
import tempfile
from types import SimpleNamespace
Expand Down Expand Up @@ -67,7 +66,7 @@ def _get_route(app, path: str):
return None


def test_metrics_and_config_routes():
def test_metrics_route():
with (
patch("fastdeploy.utils.FlexibleArgumentParser.parse_args") as mock_parse_args,
patch("fastdeploy.utils.retrive_model_from_server") as mock_retrive_model,
Expand All @@ -84,7 +83,6 @@ def test_metrics_and_config_routes():

api_server = importlib.reload(api_server_mod)

# 1) /metrics
from fastdeploy.metrics import metrics as metrics_mod

if not hasattr(metrics_mod.main_process_metrics, "cache_config_info"):
Expand All @@ -100,89 +98,6 @@ def test_metrics_and_config_routes():
)
assert "fastdeploy:" in metrics_text

# 2) /config-info
# Inject a fake engine so /config-info returns 200
from types import SimpleNamespace as NS

api_server.llm_engine = NS(cfg=NS(dummy="value"))

cfg_route = _get_route(api_server.app, "/config-info")
assert cfg_route is not None

cfg_resp = cfg_route.endpoint()
assert cfg_resp.status_code == 200
assert getattr(cfg_resp, "media_type", "").startswith("application/json")
cfg_text = (
cfg_resp.body.decode("utf-8") if isinstance(cfg_resp.body, (bytes, bytearray)) else str(cfg_resp.body)
)
data = json.loads(cfg_text)
assert isinstance(data, dict)
assert "env_config" in data


def test_config_info_engine_not_loaded_returns_500():
# Ensure we take the branch where llm_engine is None
with (
patch("fastdeploy.utils.FlexibleArgumentParser.parse_args") as mock_parse_args,
patch("fastdeploy.utils.retrive_model_from_server") as mock_retrive_model,
patch("fastdeploy.entrypoints.chat_utils.load_chat_template") as mock_load_template,
):
mock_parse_args.return_value = _build_mock_args()
mock_retrive_model.return_value = "test-model"
mock_load_template.return_value = None

from fastdeploy.entrypoints.openai import api_server as api_server_mod

api_server = importlib.reload(api_server_mod)

# Fresh import sets llm_engine to None
cfg_route = _get_route(api_server.app, "/config-info")
assert cfg_route is not None

resp = cfg_route.endpoint()
assert resp.status_code == 500
# message body is simple text
assert b"Engine not loaded" in getattr(resp, "body", b"")


def test_config_info_process_object_branches():
# Cover forcing json default() to handle
# both an object with __dict__ and one without.
with (
patch("fastdeploy.utils.FlexibleArgumentParser.parse_args") as mock_parse_args,
patch("fastdeploy.utils.retrive_model_from_server") as mock_retrive_model,
patch("fastdeploy.entrypoints.chat_utils.load_chat_template") as mock_load_template,
):
mock_parse_args.return_value = _build_mock_args()
mock_retrive_model.return_value = "test-model"
mock_load_template.return_value = None

from fastdeploy.entrypoints.openai import api_server as api_server_mod

api_server = importlib.reload(api_server_mod)

# Build a cfg with values that exercise both branches of process_object()
class WithDict:
pass

has_dict = WithDict()
has_dict.a = 1
no_dict = object()

from types import SimpleNamespace as NS

api_server.llm_engine = NS(cfg=NS(with_dict=has_dict, without_dict=no_dict))

cfg_route = _get_route(api_server.app, "/config-info")
assert cfg_route is not None

resp = cfg_route.endpoint()
assert resp.status_code == 200
data = json.loads(resp.body.decode("utf-8"))
# The object with __dict__ becomes its dict; the one without becomes null
assert data.get("with_dict") == {"a": 1}
assert "without_dict" in data and data["without_dict"] is None


def test_metrics_app_routes_when_metrics_port_diff():
# Cover metrics_app '/metrics'
Expand All @@ -208,45 +123,3 @@ def test_metrics_app_routes_when_metrics_port_diff():
assert getattr(resp, "media_type", "").startswith("text/plain")
text = resp.body.decode("utf-8") if isinstance(resp.body, (bytes, bytearray)) else str(resp.body)
assert "fastdeploy:" in text


def test_metrics_app_config_info_branches():
# Cover metrics_app '/config-info' 500 branch and success path
# including process_object branches and response
with (
patch("fastdeploy.utils.FlexibleArgumentParser.parse_args") as mock_parse_args,
patch("fastdeploy.utils.retrive_model_from_server") as mock_retrive_model,
patch("fastdeploy.entrypoints.chat_utils.load_chat_template") as mock_load_template,
):
mock_parse_args.return_value = _build_mock_args_with_side_metrics()
mock_retrive_model.return_value = "test-model"
mock_load_template.return_value = None

from fastdeploy.entrypoints.openai import api_server as api_server_mod

api_server = importlib.reload(api_server_mod)

# First, llm_engine is None -> 500
cfg_route = _get_route(api_server.metrics_app, "/config-info")
assert cfg_route is not None
resp = cfg_route.endpoint()
assert resp.status_code == 500

# Then set a fake engine with cfg carrying both serializable and non-serializable objects
class WithDict:
pass

has_dict = WithDict()
has_dict.x = 42
no_dict = object()

from types import SimpleNamespace as NS

api_server.llm_engine = NS(cfg=NS(with_dict=has_dict, without_dict=no_dict))

resp2 = cfg_route.endpoint()
assert resp2.status_code == 200
data = json.loads(resp2.body.decode("utf-8"))
assert data.get("with_dict") == {"x": 42}
assert "without_dict" in data and data["without_dict"] is None
assert "env_config" in data
Loading