mirror of
https://github.com/PaddlePaddle/FastDeploy.git
synced 2025-10-27 10:30:34 +08:00
[feat] support prefix cache clearing when /clear_load_weight is called (#4008)
* [feat] support clearing prefix cache (cherry-picked from release/2.1)
* [fix] fix ipc suffix, use port instead
* [fix] fix prefix caching not enabled
* [fix] fix key/value_cache_scales indent
* [fix] fix ep group all-reduce
* [fix] fix clear/update lock not working when workers > 1
* [chore] add preemption triggered info log
* [fix] fix code style
* [fix] fix max_num_seqs config
* [fix] do not force enable_prefix_caching=False in dynamic loading
* [fix] fix ci
* Revert "[fix] fix ci"
This reverts commit 0bc6d55cc8.
* [fix] initialize available_gpu_block_num with max_gpu_block_num
* [fix] fix config splitwise_role
* [fix] fix clearing caches synchronization and add more logs
* [chore] print cache_ready_signal in log
* [fix] fix scheduler_config.splitwise_role
* [fix] fix cache_messager cache_ready_signal create=True
* [fix] stop cache messager from launching in mixed deployment
This commit is contained in:
@@ -17,15 +17,25 @@
|
||||
from .engine_cache_queue import EngineCacheQueue
|
||||
from .engine_worker_queue import EngineWorkerQueue
|
||||
from .ipc_signal import IPCSignal, shared_memory_exists
|
||||
from .ipc_signal_const import (
|
||||
ExistTaskStatus,
|
||||
KVCacheStatus,
|
||||
ModelWeightsStatus,
|
||||
PrefixTreeStatus,
|
||||
)
|
||||
from .zmq_client import ZmqIpcClient
|
||||
from .zmq_server import ZmqIpcServer, ZmqTcpServer
|
||||
|
||||
__all__ = [
|
||||
"ZmqIpcClient",
|
||||
"ZmqIpcServer",
|
||||
"ZmqTcpServer",
|
||||
"IPCSignal",
|
||||
"EngineWorkerQueue",
|
||||
"EngineCacheQueue",
|
||||
"ZmqTcpServer",
|
||||
"ZmqIpcServer",
|
||||
"shared_memory_exists",
|
||||
"ExistTaskStatus",
|
||||
"PrefixTreeStatus",
|
||||
"ModelWeightsStatus",
|
||||
"KVCacheStatus",
|
||||
]
|
||||
|
||||
Reference in New Issue
Block a user