[misc] hide best_of from engine (#9261)
Co-authored-by: Brendan Wong <bjwpokemon@gmail.com>
This commit is contained in:
@@ -70,7 +70,6 @@ EXPECTED_VALUES = {
|
||||
[("_sum", _NUM_REQUESTS * _NUM_GENERATION_TOKENS_PER_REQUEST),
|
||||
("_count", _NUM_REQUESTS)],
|
||||
"vllm:request_params_n": [("_count", _NUM_REQUESTS)],
|
||||
"vllm:request_params_best_of": [("_count", _NUM_REQUESTS)],
|
||||
"vllm:prompt_tokens": [("_total",
|
||||
_NUM_REQUESTS * _NUM_PROMPT_TOKENS_PER_REQUEST)],
|
||||
"vllm:generation_tokens":
|
||||
@@ -151,9 +150,6 @@ EXPECTED_METRICS = [
|
||||
"vllm:request_params_n_sum",
|
||||
"vllm:request_params_n_bucket",
|
||||
"vllm:request_params_n_count",
|
||||
"vllm:request_params_best_of_sum",
|
||||
"vllm:request_params_best_of_bucket",
|
||||
"vllm:request_params_best_of_count",
|
||||
"vllm:num_preemptions_total",
|
||||
"vllm:prompt_tokens_total",
|
||||
"vllm:generation_tokens_total",
|
||||
|
||||
@@ -326,7 +326,6 @@ def assert_metrics(engine: LLMEngine, disable_log_stats: bool,
|
||||
"vllm:e2e_request_latency_seconds",
|
||||
"vllm:request_prompt_tokens",
|
||||
"vllm:request_generation_tokens",
|
||||
"vllm:request_params_best_of",
|
||||
"vllm:request_params_n",
|
||||
]
|
||||
for metric_name in request_histogram_metrics:
|
||||
|
||||
@@ -98,8 +98,6 @@ def test_traces(trace_service):
|
||||
SpanAttributes.LLM_REQUEST_TOP_P) == sampling_params.top_p
|
||||
assert attributes.get(
|
||||
SpanAttributes.LLM_REQUEST_MAX_TOKENS) == sampling_params.max_tokens
|
||||
assert attributes.get(
|
||||
SpanAttributes.LLM_REQUEST_BEST_OF) == sampling_params.best_of
|
||||
assert attributes.get(SpanAttributes.LLM_REQUEST_N) == sampling_params.n
|
||||
assert attributes.get(SpanAttributes.LLM_USAGE_PROMPT_TOKENS) == len(
|
||||
outputs[0].prompt_token_ids)
|
||||
@@ -155,8 +153,6 @@ def test_traces_with_detailed_steps(trace_service):
|
||||
SpanAttributes.LLM_REQUEST_TOP_P) == sampling_params.top_p
|
||||
assert attributes.get(
|
||||
SpanAttributes.LLM_REQUEST_MAX_TOKENS) == sampling_params.max_tokens
|
||||
assert attributes.get(
|
||||
SpanAttributes.LLM_REQUEST_BEST_OF) == sampling_params.best_of
|
||||
assert attributes.get(SpanAttributes.LLM_REQUEST_N) == sampling_params.n
|
||||
assert attributes.get(SpanAttributes.LLM_USAGE_PROMPT_TOKENS) == len(
|
||||
outputs[0].prompt_token_ids)
|
||||
|
||||
Reference in New Issue
Block a user