acdir-llada-math500 / lmdeploy /tests /pytorch /engine /test_cache_engine.py
NYCU-MLLab's picture
Upload folder using huggingface_hub
4a28d4d verified
Raw
History Blame Contribute Delete
1.69 kB
# Copyright (c) OpenMMLab. All rights reserved.
import asyncio
import pytest
from lmdeploy.pytorch.config import CacheConfig
from lmdeploy.pytorch.disagg.conn.protocol import MigrationProtocol
from lmdeploy.pytorch.disagg.messages import MigrationExecutionBatch
from lmdeploy.pytorch.engine.cache_engine import CacheEngine
def test_allocate_caches_requires_block_size_divisible_by_kernel_block_size():
cache_config = CacheConfig(max_batches=1,
block_size=96,
kernel_block_size=64,
num_cpu_blocks=0,
num_gpu_blocks=0)
with pytest.raises(ValueError, match='block_size 96 must be divisible by kernel_block_size 64'):
CacheEngine.allocate_caches(num_blocks=1,
model_config=None,
cache_config=cache_config,
world_size=1,
device='meta')
def test_pd_migration_rejects_split_kernel_blocks():
cache_engine = object.__new__(CacheEngine)
cache_engine.cache_config = CacheConfig(max_batches=1,
block_size=96,
kernel_block_size=64,
num_cpu_blocks=0,
num_gpu_blocks=0)
migration_inputs = MigrationExecutionBatch(protocol=MigrationProtocol.RDMA, requests=[])
with pytest.raises(RuntimeError, match='PD migration does not support block_size != kernel_block_size'):
asyncio.run(cache_engine.migrate(migration_inputs))