[Model][6/N] Improve all pooling task | Support chunked prefill with ALL pooling (#27145)
Signed-off-by: wang.yuqi <noooop@126.com> Signed-off-by: wang.yuqi <yuqi.wang@daocloud.io> Co-authored-by: gemini-code-assist[bot] <176961590+gemini-code-assist[bot]@users.noreply.github.com>
This commit is contained in:
@@ -64,7 +64,7 @@ from vllm.multimodal.profiling import BaseDummyInputsBuilder
|
||||
from vllm.sequence import IntermediateTensors
|
||||
|
||||
from .interfaces import IsAttentionFree, MultiModalEmbeddings, SupportsMultiModal
|
||||
from .interfaces_base import default_pooling_type
|
||||
from .interfaces_base import attn_type
|
||||
|
||||
logger = init_logger(__name__)
|
||||
|
||||
@@ -220,7 +220,7 @@ class TerratorchMultiModalProcessor(BaseMultiModalProcessor):
|
||||
)
|
||||
|
||||
|
||||
@default_pooling_type("All")
|
||||
@attn_type("attention_free")
|
||||
@MULTIMODAL_REGISTRY.register_processor(
|
||||
TerratorchMultiModalProcessor,
|
||||
info=TerratorchProcessingInfo,
|
||||
|
||||
Reference in New Issue
Block a user