[Frontend] Log the maximum supported concurrency (#8831)

0b5b5d76 · AlpinDale · GitHub · cdc72e3c · 0b5b5d76 · 0b5b5d76
Unverified Commit 0b5b5d76 authored Oct 09, 2024 by AlpinDale Committed by GitHub Oct 09, 2024
Hide whitespace changes
Inline Side-by-side

Showing with 8 additions and 0 deletions

vllm/executor/distributed_gpu_executor.py vllm/executor/distributed_gpu_executor.py +4 -0

vllm/executor/gpu_executor.py vllm/executor/gpu_executor.py +4 -0

No files found.
--- a/vllm/executor/distributed_gpu_executor.py
+++ b/vllm/executor/distributed_gpu_executor.py
@@ -56,6 +56,10 @@ class DistributedGPUExecutor(GPUExecutor):
        # have GPUs.
        logger.info("# GPU blocks: %d, # CPU blocks: %d", num_gpu_blocks,
                    num_cpu_blocks)
+        max_concurrency = (num_gpu_blocks * self.cache_config.block_size /
+                           self.model_config.max_model_len)
+        logger.info("Maximum concurrency for %s tokens per request: %.2fx",
+                    self.model_config.max_model_len, max_concurrency)
        self.cache_config.num_gpu_blocks = num_gpu_blocks
        self.cache_config.num_cpu_blocks = num_cpu_blocks

--- a/vllm/executor/gpu_executor.py
+++ b/vllm/executor/gpu_executor.py
@@ -121,6 +121,10 @@ class GPUExecutor(ExecutorBase):
        # remains to abstract away the device for non-GPU configurations.
        logger.info("# GPU blocks: %d, # CPU blocks: %d", num_gpu_blocks,
                    num_cpu_blocks)
+        max_concurrency = (num_gpu_blocks * self.cache_config.block_size /
+                           self.model_config.max_model_len)
+        logger.info("Maximum concurrency for %s tokens per request: %.2fx",
+                    self.model_config.max_model_len, max_concurrency)
        self.driver_worker.initialize_cache(num_gpu_blocks, num_cpu_blocks)