[Model] Pooling models default to using chunked prefill & prefix caching if supported. (#20930)
Signed-off-by: wang.yuqi <noooop@126.com>
This commit is contained in:
@@ -2,73 +2,78 @@
|
||||
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
||||
import pytest
|
||||
|
||||
from ...utils import EmbedModelInfo, RerankModelInfo
|
||||
from ...utils import (CLSPoolingEmbedModelInfo, CLSPoolingRerankModelInfo,
|
||||
EmbedModelInfo, LASTPoolingEmbedModelInfo,
|
||||
RerankModelInfo)
|
||||
from .embed_utils import correctness_test_embed_models
|
||||
from .mteb_utils import mteb_test_embed_models, mteb_test_rerank_models
|
||||
|
||||
MODELS = [
|
||||
########## BertModel
|
||||
EmbedModelInfo("BAAI/bge-base-en",
|
||||
architecture="BertModel",
|
||||
enable_test=True),
|
||||
EmbedModelInfo("BAAI/bge-base-zh",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
EmbedModelInfo("BAAI/bge-small-en",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
EmbedModelInfo("BAAI/bge-small-zh",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
EmbedModelInfo("BAAI/bge-large-en",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
EmbedModelInfo("BAAI/bge-large-zh",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
EmbedModelInfo("BAAI/bge-large-zh-noinstruct",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
EmbedModelInfo("BAAI/bge-base-en-v1.5",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
EmbedModelInfo("BAAI/bge-base-zh-v1.5",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
EmbedModelInfo("BAAI/bge-small-en-v1.5",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
EmbedModelInfo("BAAI/bge-small-zh-v1.5",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
EmbedModelInfo("BAAI/bge-large-en-v1.5",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
EmbedModelInfo("BAAI/bge-large-zh-v1.5",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-base-en",
|
||||
architecture="BertModel",
|
||||
enable_test=True),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-base-zh",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-small-en",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-small-zh",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-large-en",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-large-zh",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-large-zh-noinstruct",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-base-en-v1.5",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-base-zh-v1.5",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-small-en-v1.5",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-small-zh-v1.5",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-large-en-v1.5",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-large-zh-v1.5",
|
||||
architecture="BertModel",
|
||||
enable_test=False),
|
||||
########## XLMRobertaModel
|
||||
EmbedModelInfo("BAAI/bge-m3",
|
||||
architecture="XLMRobertaModel",
|
||||
enable_test=True),
|
||||
CLSPoolingEmbedModelInfo("BAAI/bge-m3",
|
||||
architecture="XLMRobertaModel",
|
||||
enable_test=True),
|
||||
########## Qwen2Model
|
||||
EmbedModelInfo("BAAI/bge-code-v1",
|
||||
architecture="Qwen2Model",
|
||||
dtype="float32",
|
||||
enable_test=True),
|
||||
LASTPoolingEmbedModelInfo("BAAI/bge-code-v1",
|
||||
architecture="Qwen2Model",
|
||||
dtype="float32",
|
||||
enable_test=True),
|
||||
]
|
||||
|
||||
RERANK_MODELS = [
|
||||
########## XLMRobertaForSequenceClassification
|
||||
RerankModelInfo("BAAI/bge-reranker-base",
|
||||
architecture="XLMRobertaForSequenceClassification",
|
||||
enable_test=True),
|
||||
RerankModelInfo("BAAI/bge-reranker-large",
|
||||
architecture="XLMRobertaForSequenceClassification",
|
||||
enable_test=False),
|
||||
RerankModelInfo("BAAI/bge-reranker-v2-m3",
|
||||
architecture="XLMRobertaForSequenceClassification",
|
||||
enable_test=False)
|
||||
CLSPoolingRerankModelInfo(
|
||||
"BAAI/bge-reranker-base",
|
||||
architecture="XLMRobertaForSequenceClassification",
|
||||
enable_test=True),
|
||||
CLSPoolingRerankModelInfo(
|
||||
"BAAI/bge-reranker-large",
|
||||
architecture="XLMRobertaForSequenceClassification",
|
||||
enable_test=False),
|
||||
CLSPoolingRerankModelInfo(
|
||||
"BAAI/bge-reranker-v2-m3",
|
||||
architecture="XLMRobertaForSequenceClassification",
|
||||
enable_test=False)
|
||||
]
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user