[Core] Encoder separation for Encode-Prefill-Decode Disaggregation (#25233)
Signed-off-by: n00909098 <nguyen.kha.long@huawei.com> Signed-off-by: knlnguyen1802 <knlnguyen1802@gmail.com> Signed-off-by: herotai214 <herotai214@gmail.com> Signed-off-by: Khuong Le <khuong.le.manh@huawei.com> Signed-off-by: Khuong Le <lemanhkhuong2611@gmail.com> Co-authored-by: n00909098 <nguyen.kha.long@huawei.com> Co-authored-by: knlnguyen1802 <knlnguyen1802@gmail.com> Co-authored-by: herotai214 <herotai214@gmail.com> Co-authored-by: gemini-code-assist[bot] <176961590+gemini-code-assist[bot]@users.noreply.github.com> Co-authored-by: Khuong Le <khuong.le.manh@huawei.com> Co-authored-by: Khuong Le <lemanhkhuong2611@gmail.com>
This commit is contained in:
@@ -49,10 +49,18 @@ def kernel_warmup(worker: "Worker"):
|
||||
except NotImplementedError:
|
||||
return False
|
||||
|
||||
if not worker.model_runner.is_pooling_model and all(
|
||||
_is_flashinfer_backend(group.backend)
|
||||
for groups in worker.model_runner.attn_groups
|
||||
for group in groups
|
||||
# NOTE: we add check for empty attn_groups to avoid errors when
|
||||
# deploying models such as E instances and encoder-only models.
|
||||
# As for those models, worker.model_runner.attn_groups is empty.
|
||||
# This change is made during EPD feature development.
|
||||
if (
|
||||
not worker.model_runner.is_pooling_model
|
||||
and worker.model_runner.attn_groups
|
||||
and all(
|
||||
_is_flashinfer_backend(group.backend)
|
||||
for groups in worker.model_runner.attn_groups
|
||||
for group in groups
|
||||
)
|
||||
):
|
||||
logger.info("Warming up FlashInfer attention.")
|
||||
# Warmup with mixed batch containing both prefill and decode tokens
|
||||
|
||||
Reference in New Issue
Block a user