vllm/vllm/utils.py

import enum
import os
import socket
import subprocess
import uuid
import gc
from platform import uname
from typing import List, Tuple, Union
from packaging.version import parse, Version

import psutil
import torch
import asyncio
from functools import partial
from typing import (
    Awaitable,
    Callable,
    TypeVar,
)
from collections import OrderedDict
from typing import Any, Hashable, Optional

from vllm.logger import init_logger

T = TypeVar("T")
logger = init_logger(__name__)

STR_DTYPE_TO_TORCH_DTYPE = {
    "half": torch.half,
    "bfloat16": torch.bfloat16,
    "float": torch.float,
    "fp8_e5m2": torch.uint8,
}


class Device(enum.Enum):
    GPU = enum.auto()
    CPU = enum.auto()


class Counter:

    def __init__(self, start: int = 0) -> None:
        self.counter = start

    def __next__(self) -> int:
        i = self.counter
        self.counter += 1
        return i

    def reset(self) -> None:
        self.counter = 0


class LRUCache:

    def __init__(self, capacity: int):
        self.cache = OrderedDict()
        self.capacity = capacity

    def __contains__(self, key: Hashable) -> bool:
        return key in self.cache

    def __len__(self) -> int:
        return len(self.cache)

    def __getitem__(self, key: Hashable) -> Any:
        return self.get(key)

    def __setitem__(self, key: Hashable, value: Any) -> None:
        self.put(key, value)

    def __delitem__(self, key: Hashable) -> None:
        self.pop(key)

    def touch(self, key: Hashable) -> None:
        self.cache.move_to_end(key)

    def get(self, key: Hashable, default_value: Optional[Any] = None) -> int:
        if key in self.cache:
            value = self.cache[key]
            self.cache.move_to_end(key)
        else:
            value = default_value
        return value

    def put(self, key: Hashable, value: Any) -> None:
        self.cache[key] = value
        self.cache.move_to_end(key)
        self._remove_old_if_needed()

    def _on_remove(self, key: Hashable, value: Any):
        pass

    def remove_oldest(self):
        if not self.cache:
            return
        key, value = self.cache.popitem(last=False)
        self._on_remove(key, value)

    def _remove_old_if_needed(self) -> None:
        while len(self.cache) > self.capacity:
            self.remove_oldest()

    def pop(self, key: int, default_value: Optional[Any] = None) -> Any:
        run_on_remove = key in self.cache
        value = self.cache.pop(key, default_value)
        if run_on_remove:
            self._on_remove(key, value)
        return value

    def clear(self):
        while len(self.cache) > 0:
            self.remove_oldest()
        self.cache.clear()


def is_hip() -> bool:
    return torch.version.hip is not None


def is_neuron() -> bool:
    try:
        import transformers_neuronx
    except ImportError:
        transformers_neuronx = None
    return transformers_neuronx is not None


def get_max_shared_memory_bytes(gpu: int = 0) -> int:
    """Returns the maximum shared memory per thread block in bytes."""
    # NOTE: This import statement should be executed lazily since
    # the Neuron-X backend does not have the `cuda_utils` module.
    from vllm._C import cuda_utils

    max_shared_mem = (
        cuda_utils.get_max_shared_memory_per_block_device_attribute(gpu))
    # value 0 will cause MAX_SEQ_LEN become negative and test_attention.py
    # will fail
    assert max_shared_mem > 0, "max_shared_mem can not be zero"
    return int(max_shared_mem)


def get_cpu_memory() -> int:
    """Returns the total CPU memory of the node in bytes."""
    return psutil.virtual_memory().total


def random_uuid() -> str:
    return str(uuid.uuid4().hex)


def in_wsl() -> bool:
    # Reference: https://github.com/microsoft/WSL/issues/4071
    return "microsoft" in " ".join(uname()).lower()


def make_async(func: Callable[..., T]) -> Callable[..., Awaitable[T]]:
    """Take a blocking function, and run it on in an executor thread.

    This function prevents the blocking function from blocking the
    asyncio event loop.
    The code in this function needs to be thread safe.
    """

    def _async_wrapper(*args, **kwargs) -> asyncio.Future:
        loop = asyncio.get_event_loop()
        p_func = partial(func, *args, **kwargs)
        return loop.run_in_executor(executor=None, func=p_func)

    return _async_wrapper


def get_ip() -> str:
    # try ipv4
    s = socket.socket(socket.AF_INET, socket.SOCK_DGRAM)
    try:
        s.connect(("8.8.8.8", 80))  # Doesn't need to be reachable
        return s.getsockname()[0]
    except OSError:
        # try ipv6
        s = socket.socket(socket.AF_INET6, socket.SOCK_DGRAM)
        s.connect(("dns.google", 80))
        return s.getsockname()[0]


def get_distributed_init_method(ip: str, port: int) -> str:
    return f"tcp://{ip}:{port}"


def get_open_port() -> int:
    # try ipv4
    try:
        with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
            s.bind(("", 0))
            return s.getsockname()[1]
    except OSError:
        # try ipv6
        with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s:
            s.bind(("", 0))
            return s.getsockname()[1]


def set_cuda_visible_devices(device_ids: List[int]) -> None:
    os.environ["CUDA_VISIBLE_DEVICES"] = ",".join(map(str, device_ids))


def get_nvcc_cuda_version() -> Optional[Version]:
    cuda_home = os.environ.get('CUDA_HOME')
    if not cuda_home:
        cuda_home = '/usr/local/cuda'
        if os.path.isfile(cuda_home + '/bin/nvcc'):
            logger.info(f'CUDA_HOME is not found in the environment. '
                        f'Using {cuda_home} as CUDA_HOME.')
        else:
            logger.warning(
                f'Not found nvcc in {cuda_home}. Skip cuda version check!')
            return None
    nvcc_output = subprocess.check_output([cuda_home + "/bin/nvcc", "-V"],
                                          universal_newlines=True)
    output = nvcc_output.split()
    release_idx = output.index("release") + 1
    nvcc_cuda_version = parse(output[release_idx].split(",")[0])
    return nvcc_cuda_version


def _generate_random_fp8_e5m2(
    tensor: torch.tensor,
    low: float,
    high: float,
) -> None:
    # NOTE(zhaoyang): Due to NaN and Inf representation for fp8 data type,
    # it may occur Inf or NaN if we directly use torch.randint
    # to generate random data for fp8 data.
    # For example, s.11111.00 in fp8e5m2 format represents Inf.
    #     | E4M3        | E5M2
    #-----|-------------|-------------------
    # Inf | N/A         | s.11111.00
    # NaN | s.1111.111  | s.11111.{01,10,11}
    from vllm._C import cache_ops
    tensor_tmp = torch.empty_like(tensor, dtype=torch.float16)
    tensor_tmp.uniform_(low, high)
    cache_ops.convert_fp8_e5m2(tensor_tmp, tensor)
    del tensor_tmp


def create_kv_caches_with_random(
    num_blocks: int,
    block_size: int,
    num_layers: int,
    num_heads: int,
    head_size: int,
    cache_dtype: Optional[Union[str, torch.dtype]],
    model_dtype: Optional[Union[str, torch.dtype]] = None,
    seed: Optional[int] = 0,
    device: Optional[str] = "cuda",
) -> Tuple[List[torch.Tensor], List[torch.Tensor]]:
    torch.random.manual_seed(seed)
    if torch.cuda.is_available():
        torch.cuda.manual_seed(seed)

    if isinstance(cache_dtype, str):
        if cache_dtype == "auto":
            if isinstance(model_dtype, str):
                torch_dtype = STR_DTYPE_TO_TORCH_DTYPE[model_dtype]
            elif isinstance(model_dtype, torch.dtype):
                torch_dtype = model_dtype
            else:
                raise ValueError(f"Invalid model dtype: {model_dtype}")
        elif cache_dtype in ["half", "bfloat16", "float"]:
            torch_dtype = STR_DTYPE_TO_TORCH_DTYPE[cache_dtype]
        elif cache_dtype == "fp8_e5m2":
            torch_dtype = torch.uint8
        else:
            raise ValueError(f"Invalid kv cache dtype: {cache_dtype}")
    elif isinstance(cache_dtype, torch.dtype):
        torch_dtype = cache_dtype
    else:
        raise ValueError(f"Invalid kv cache dtype: {cache_dtype}")

    scale = head_size**-0.5
    x = 16 // torch.tensor([], dtype=torch_dtype).element_size()
    key_cache_shape = (num_blocks, num_heads, head_size // x, block_size, x)
    key_caches = []
    for _ in range(num_layers):
        key_cache = torch.empty(size=key_cache_shape,
                                dtype=torch_dtype,
                                device=device)
        if cache_dtype == 'fp8_e5m2':
            _generate_random_fp8_e5m2(key_cache, -scale, scale)
        elif torch_dtype in [torch.half, torch.bfloat16, torch.float]:
            key_cache.uniform_(-scale, scale)
        else:
            raise ValueError(
                f"Does not support key cache of type {cache_dtype}")
        key_caches.append(key_cache)

    value_cache_shape = (num_blocks, num_heads, head_size, block_size)
    value_caches = []
    for _ in range(num_layers):
        value_cache = torch.empty(size=value_cache_shape,
                                  dtype=torch_dtype,
                                  device=device)
        if cache_dtype == 'fp8_e5m2':
            _generate_random_fp8_e5m2(value_cache, -scale, scale)
        elif torch_dtype in [torch.half, torch.bfloat16, torch.float]:
            value_cache.uniform_(-scale, scale)
        else:
            raise ValueError(
                f"Does not support value cache of type {cache_dtype}")
        value_caches.append(value_cache)
    return key_caches, value_caches


class measure_cuda_memory:

    def __init__(self, device=None):
        self.device = device

    def current_memory_usage(self) -> float:
        # Return the memory usage in bytes.
        torch.cuda.reset_peak_memory_stats(self.device)
        mem = torch.cuda.max_memory_allocated(self.device)
        return mem

    def __enter__(self):
        self.initial_memory = self.current_memory_usage()
        # This allows us to call methods of the context manager if needed
        return self

    def __exit__(self, exc_type, exc_val, exc_tb):
        self.final_memory = self.current_memory_usage()
        self.consumed_memory = self.final_memory - self.initial_memory

        # Force garbage collection
        gc.collect()
Add utils 2023-02-09 11:26:50 +00:00			`import enum`
Use NCCL instead of ray for control-plane communication to remove serialization overhead (#2221) 2024-01-04 03:30:22 +08:00			`import os`
Optimize model execution with CUDA graph (#1926) Co-authored-by: Chen Shen <scv119@gmail.com> Co-authored-by: Antoni Baum <antoni.baum@protonmail.com> 2023-12-16 21:12:08 -08:00			`import socket`
Support FP8-E5M2 KV Cache (#2279) Co-authored-by: zhaoyang <zhao.yang16@zte.com.cn> Co-authored-by: Zhuohan Li <zhuohan123@gmail.com> 2024-01-29 08:43:54 +08:00			`import subprocess`
OpenAI Compatible Frontend (#116) 2023-05-23 21:39:50 -07:00			`import uuid`
Measure model memory usage (#3120) 2024-03-07 11:42:42 -08:00			`import gc`
Allocate more shared memory to attention kernel (#1154) 2023-09-26 22:27:13 -07:00			`from platform import uname`
Support FP8-E5M2 KV Cache (#2279) Co-authored-by: zhaoyang <zhao.yang16@zte.com.cn> Co-authored-by: Zhuohan Li <zhuohan123@gmail.com> 2024-01-29 08:43:54 +08:00			`from typing import List, Tuple, Union`
			`from packaging.version import parse, Version`
Support tensor parallel (#2) 2023-03-22 04:45:42 +08:00
Refactor system architecture (#82) 2023-05-09 15:30:12 -07:00			`import psutil`
Support tensor parallel (#2) 2023-03-22 04:45:42 +08:00			`import torch`
[Experimental] Add multi-LoRA support (#1804) Co-authored-by: Chen Shen <scv119@gmail.com> Co-authored-by: Shreyas Krishnaswamy <shrekris@anyscale.com> Co-authored-by: Avnish Narayan <avnish@anyscale.com> 2024-01-24 00:26:37 +01:00			`import asyncio`
			`from functools import partial`
			`from typing import (`
			`Awaitable,`
			`Callable,`
			`TypeVar,`
			`)`
			`from collections import OrderedDict`
			`from typing import Any, Hashable, Optional`

Support FP8-E5M2 KV Cache (#2279) Co-authored-by: zhaoyang <zhao.yang16@zte.com.cn> Co-authored-by: Zhuohan Li <zhuohan123@gmail.com> 2024-01-29 08:43:54 +08:00			`from vllm.logger import init_logger`

[Experimental] Add multi-LoRA support (#1804) Co-authored-by: Chen Shen <scv119@gmail.com> Co-authored-by: Shreyas Krishnaswamy <shrekris@anyscale.com> Co-authored-by: Avnish Narayan <avnish@anyscale.com> 2024-01-24 00:26:37 +01:00			`T = TypeVar("T")`
Support FP8-E5M2 KV Cache (#2279) Co-authored-by: zhaoyang <zhao.yang16@zte.com.cn> Co-authored-by: Zhuohan Li <zhuohan123@gmail.com> 2024-01-29 08:43:54 +08:00			`logger = init_logger(__name__)`

			`STR_DTYPE_TO_TORCH_DTYPE = {`
			`"half": torch.half,`
			`"bfloat16": torch.bfloat16,`
			`"float": torch.float,`
			`"fp8_e5m2": torch.uint8,`
			`}`
Support tensor parallel (#2) 2023-03-22 04:45:42 +08:00
Add utils 2023-02-09 11:26:50 +00:00
			`class Device(enum.Enum):`
			`GPU = enum.auto()`
			`CPU = enum.auto()`


			`class Counter:`

			`def __init__(self, start: int = 0) -> None:`
			`self.counter = start`

Fix typo 2023-02-14 01:19:27 +00:00			`def __next__(self) -> int:`
[Quality] Add code formatter and linter (#326) 2023-07-03 11:31:55 -07:00			`i = self.counter`
Add utils 2023-02-09 11:26:50 +00:00			`self.counter += 1`
[Quality] Add code formatter and linter (#326) 2023-07-03 11:31:55 -07:00			`return i`
Add utils 2023-02-09 11:26:50 +00:00
			`def reset(self) -> None:`
			`self.counter = 0`
Support tensor parallel (#2) 2023-03-22 04:45:42 +08:00
FastAPI-based working frontend (#10) 2023-03-29 14:48:56 +08:00
[Experimental] Add multi-LoRA support (#1804) Co-authored-by: Chen Shen <scv119@gmail.com> Co-authored-by: Shreyas Krishnaswamy <shrekris@anyscale.com> Co-authored-by: Avnish Narayan <avnish@anyscale.com> 2024-01-24 00:26:37 +01:00			`class LRUCache:`

			`def __init__(self, capacity: int):`
			`self.cache = OrderedDict()`
			`self.capacity = capacity`

			`def __contains__(self, key: Hashable) -> bool:`
			`return key in self.cache`

			`def __len__(self) -> int:`
			`return len(self.cache)`

			`def __getitem__(self, key: Hashable) -> Any:`
			`return self.get(key)`

			`def __setitem__(self, key: Hashable, value: Any) -> None:`
			`self.put(key, value)`

			`def __delitem__(self, key: Hashable) -> None:`
			`self.pop(key)`

			`def touch(self, key: Hashable) -> None:`
			`self.cache.move_to_end(key)`

			`def get(self, key: Hashable, default_value: Optional[Any] = None) -> int:`
			`if key in self.cache:`
			`value = self.cache[key]`
			`self.cache.move_to_end(key)`
			`else:`
			`value = default_value`
			`return value`

			`def put(self, key: Hashable, value: Any) -> None:`
			`self.cache[key] = value`
			`self.cache.move_to_end(key)`
			`self._remove_old_if_needed()`

			`def _on_remove(self, key: Hashable, value: Any):`
			`pass`

			`def remove_oldest(self):`
			`if not self.cache:`
			`return`
			`key, value = self.cache.popitem(last=False)`
			`self._on_remove(key, value)`

			`def _remove_old_if_needed(self) -> None:`
			`while len(self.cache) > self.capacity:`
			`self.remove_oldest()`

			`def pop(self, key: int, default_value: Optional[Any] = None) -> Any:`
			`run_on_remove = key in self.cache`
			`value = self.cache.pop(key, default_value)`
			`if run_on_remove:`
			`self._on_remove(key, value)`
			`return value`

			`def clear(self):`
			`while len(self.cache) > 0:`
			`self.remove_oldest()`
			`self.cache.clear()`


Merge EmbeddedLLM/vllm-rocm into vLLM main (#1836) Co-authored-by: Philipp Moritz <pcmoritz@gmail.com> Co-authored-by: Amir Balwel <amoooori04@gmail.com> Co-authored-by: root <kuanfu.liu@akirakan.com> Co-authored-by: tjtanaa <tunjian.tan@embeddedllm.com> Co-authored-by: kuanfu <kuanfu.liu@embeddedllm.com> Co-authored-by: miloice <17350011+kliuae@users.noreply.github.com> 2023-12-08 15:16:52 +08:00			`def is_hip() -> bool:`
			`return torch.version.hip is not None`


[Neuron] Support inference with transformers-neuronx (#2569) 2024-02-28 09:34:34 -08:00			`def is_neuron() -> bool:`
			`try:`
			`import transformers_neuronx`
			`except ImportError:`
			`transformers_neuronx = None`
			`return transformers_neuronx is not None`


Allocate more shared memory to attention kernel (#1154) 2023-09-26 22:27:13 -07:00			`def get_max_shared_memory_bytes(gpu: int = 0) -> int:`
			`"""Returns the maximum shared memory per thread block in bytes."""`
[Neuron] Add an option to build with neuron (#2065) 2024-01-18 10:58:50 -08:00			`# NOTE: This import statement should be executed lazily since`
			# the Neuron-X backend does not have the `cuda_utils` module.
			`from vllm._C import cuda_utils`

Re-enable the 80 char line width limit (#3305) 2024-03-10 19:49:14 -07:00			`max_shared_mem = (`
			`cuda_utils.get_max_shared_memory_per_block_device_attribute(gpu))`
			`# value 0 will cause MAX_SEQ_LEN become negative and test_attention.py`
			`# will fail`
[ROCm] add support to ROCm 6.0 and MI300 (#2274) 2024-01-26 15:41:10 -05:00			`assert max_shared_mem > 0, "max_shared_mem can not be zero"`
Allocate more shared memory to attention kernel (#1154) 2023-09-26 22:27:13 -07:00			`return int(max_shared_mem)`


FastAPI-based working frontend (#10) 2023-03-29 14:48:56 +08:00			`def get_cpu_memory() -> int:`
Print warnings/errors for large swap space (#123) 2023-05-23 18:22:26 -07:00			`"""Returns the total CPU memory of the node in bytes."""`
FastAPI-based working frontend (#10) 2023-03-29 14:48:56 +08:00			`return psutil.virtual_memory().total`
OpenAI Compatible Frontend (#116) 2023-05-23 21:39:50 -07:00

			`def random_uuid() -> str:`
			`return str(uuid.uuid4().hex)`
[Fix] Do not pin memory when in WSL (#312) 2023-06-29 15:00:21 -07:00
[Quality] Add code formatter and linter (#326) 2023-07-03 11:31:55 -07:00
[Fix] Do not pin memory when in WSL (#312) 2023-06-29 15:00:21 -07:00			`def in_wsl() -> bool:`
			`# Reference: https://github.com/microsoft/WSL/issues/4071`
			`return "microsoft" in " ".join(uname()).lower()`
Optimize model execution with CUDA graph (#1926) Co-authored-by: Chen Shen <scv119@gmail.com> Co-authored-by: Antoni Baum <antoni.baum@protonmail.com> 2023-12-16 21:12:08 -08:00

[Experimental] Add multi-LoRA support (#1804) Co-authored-by: Chen Shen <scv119@gmail.com> Co-authored-by: Shreyas Krishnaswamy <shrekris@anyscale.com> Co-authored-by: Avnish Narayan <avnish@anyscale.com> 2024-01-24 00:26:37 +01:00			`def make_async(func: Callable[..., T]) -> Callable[..., Awaitable[T]]:`
			`"""Take a blocking function, and run it on in an executor thread.`

			`This function prevents the blocking function from blocking the`
			`asyncio event loop.`
			`The code in this function needs to be thread safe.`
			`"""`

			`def _async_wrapper(args, *kwargs) -> asyncio.Future:`
			`loop = asyncio.get_event_loop()`
			`p_func = partial(func, args, *kwargs)`
			`return loop.run_in_executor(executor=None, func=p_func)`

			`return _async_wrapper`


Use NCCL instead of ray for control-plane communication to remove serialization overhead (#2221) 2024-01-04 03:30:22 +08:00			`def get_ip() -> str:`
fix `get_ip` error in pure ipv6 environment (#2931) 2024-02-27 11:22:16 +08:00			`# try ipv4`
`get_ip()`: Fix ipv4 ipv6 dualstack (#2408) 2024-01-10 11:39:58 -08:00			`s = socket.socket(socket.AF_INET, socket.SOCK_DGRAM)`
fix `get_ip` error in pure ipv6 environment (#2931) 2024-02-27 11:22:16 +08:00			`try:`
[Minor fix] The domain dns.google may cause a socket.gaierror exception (#3176) Co-authored-by: guofangze <guofangze@kuaishou.com> 2024-03-05 03:17:12 +08:00			`s.connect(("8.8.8.8", 80)) # Doesn't need to be reachable`
fix `get_ip` error in pure ipv6 environment (#2931) 2024-02-27 11:22:16 +08:00			`return s.getsockname()[0]`
			`except OSError:`
			`# try ipv6`
			`s = socket.socket(socket.AF_INET6, socket.SOCK_DGRAM)`
			`s.connect(("dns.google", 80))`
			`return s.getsockname()[0]`
Use NCCL instead of ray for control-plane communication to remove serialization overhead (#2221) 2024-01-04 03:30:22 +08:00

[Speculative decoding 2/9] Multi-step worker for draft model (#2424) 2024-01-21 16:31:47 -08:00			`def get_distributed_init_method(ip: str, port: int) -> str:`
			`return f"tcp://{ip}:{port}"`


Use NCCL instead of ray for control-plane communication to remove serialization overhead (#2221) 2024-01-04 03:30:22 +08:00			`def get_open_port() -> int:`
fix `get_ip` error in pure ipv6 environment (#2931) 2024-02-27 11:22:16 +08:00			`# try ipv4`
			`try:`
			`with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:`
			`s.bind(("", 0))`
			`return s.getsockname()[1]`
			`except OSError:`
			`# try ipv6`
			`with socket.socket(socket.AF_INET6, socket.SOCK_STREAM) as s:`
			`s.bind(("", 0))`
			`return s.getsockname()[1]`
Use NCCL instead of ray for control-plane communication to remove serialization overhead (#2221) 2024-01-04 03:30:22 +08:00

			`def set_cuda_visible_devices(device_ids: List[int]) -> None:`
			`os.environ["CUDA_VISIBLE_DEVICES"] = ",".join(map(str, device_ids))`
Support FP8-E5M2 KV Cache (#2279) Co-authored-by: zhaoyang <zhao.yang16@zte.com.cn> Co-authored-by: Zhuohan Li <zhuohan123@gmail.com> 2024-01-29 08:43:54 +08:00

Fix nvcc not found in vlm-openai image (#2781) 2024-02-23 06:25:07 +08:00			`def get_nvcc_cuda_version() -> Optional[Version]:`
Support FP8-E5M2 KV Cache (#2279) Co-authored-by: zhaoyang <zhao.yang16@zte.com.cn> Co-authored-by: Zhuohan Li <zhuohan123@gmail.com> 2024-01-29 08:43:54 +08:00			`cuda_home = os.environ.get('CUDA_HOME')`
			`if not cuda_home:`
			`cuda_home = '/usr/local/cuda'`
Fix nvcc not found in vlm-openai image (#2781) 2024-02-23 06:25:07 +08:00			`if os.path.isfile(cuda_home + '/bin/nvcc'):`
Re-enable the 80 char line width limit (#3305) 2024-03-10 19:49:14 -07:00			`logger.info(f'CUDA_HOME is not found in the environment. '`
			`f'Using {cuda_home} as CUDA_HOME.')`
Fix nvcc not found in vlm-openai image (#2781) 2024-02-23 06:25:07 +08:00			`else:`
			`logger.warning(`
			`f'Not found nvcc in {cuda_home}. Skip cuda version check!')`
			`return None`
Support FP8-E5M2 KV Cache (#2279) Co-authored-by: zhaoyang <zhao.yang16@zte.com.cn> Co-authored-by: Zhuohan Li <zhuohan123@gmail.com> 2024-01-29 08:43:54 +08:00			`nvcc_output = subprocess.check_output([cuda_home + "/bin/nvcc", "-V"],`
			`universal_newlines=True)`
			`output = nvcc_output.split()`
			`release_idx = output.index("release") + 1`
			`nvcc_cuda_version = parse(output[release_idx].split(",")[0])`
			`return nvcc_cuda_version`


			`def _generate_random_fp8_e5m2(`
			`tensor: torch.tensor,`
			`low: float,`
			`high: float,`
			`) -> None:`
			`# NOTE(zhaoyang): Due to NaN and Inf representation for fp8 data type,`
			`# it may occur Inf or NaN if we directly use torch.randint`
			`# to generate random data for fp8 data.`
chore(vllm): codespell for spell checking (#2820) 2024-02-22 02:56:01 +00:00			`# For example, s.11111.00 in fp8e5m2 format represents Inf.`
Support FP8-E5M2 KV Cache (#2279) Co-authored-by: zhaoyang <zhao.yang16@zte.com.cn> Co-authored-by: Zhuohan Li <zhuohan123@gmail.com> 2024-01-29 08:43:54 +08:00			`# \| E4M3 \| E5M2`
			`#-----\|-------------\|-------------------`
			`# Inf \| N/A \| s.11111.00`
			`# NaN \| s.1111.111 \| s.11111.{01,10,11}`
			`from vllm._C import cache_ops`
			`tensor_tmp = torch.empty_like(tensor, dtype=torch.float16)`
			`tensor_tmp.uniform_(low, high)`
			`cache_ops.convert_fp8_e5m2(tensor_tmp, tensor)`
			`del tensor_tmp`


			`def create_kv_caches_with_random(`
			`num_blocks: int,`
			`block_size: int,`
			`num_layers: int,`
			`num_heads: int,`
			`head_size: int,`
			`cache_dtype: Optional[Union[str, torch.dtype]],`
			`model_dtype: Optional[Union[str, torch.dtype]] = None,`
			`seed: Optional[int] = 0,`
			`device: Optional[str] = "cuda",`
			`) -> Tuple[List[torch.Tensor], List[torch.Tensor]]:`
			`torch.random.manual_seed(seed)`
Remove hardcoded `device="cuda" ` to support more devices (#2503) Co-authored-by: Jiang Li <jiang1.li@intel.com> Co-authored-by: Kunshang Ji <kunshang.ji@intel.com> 2024-02-02 07:46:39 +08:00			`if torch.cuda.is_available():`
			`torch.cuda.manual_seed(seed)`
Support FP8-E5M2 KV Cache (#2279) Co-authored-by: zhaoyang <zhao.yang16@zte.com.cn> Co-authored-by: Zhuohan Li <zhuohan123@gmail.com> 2024-01-29 08:43:54 +08:00
			`if isinstance(cache_dtype, str):`
			`if cache_dtype == "auto":`
			`if isinstance(model_dtype, str):`
			`torch_dtype = STR_DTYPE_TO_TORCH_DTYPE[model_dtype]`
			`elif isinstance(model_dtype, torch.dtype):`
			`torch_dtype = model_dtype`
			`else:`
			`raise ValueError(f"Invalid model dtype: {model_dtype}")`
			`elif cache_dtype in ["half", "bfloat16", "float"]:`
			`torch_dtype = STR_DTYPE_TO_TORCH_DTYPE[cache_dtype]`
			`elif cache_dtype == "fp8_e5m2":`
			`torch_dtype = torch.uint8`
			`else:`
			`raise ValueError(f"Invalid kv cache dtype: {cache_dtype}")`
			`elif isinstance(cache_dtype, torch.dtype):`
			`torch_dtype = cache_dtype`
			`else:`
			`raise ValueError(f"Invalid kv cache dtype: {cache_dtype}")`

			`scale = head_size**-0.5`
			`x = 16 // torch.tensor([], dtype=torch_dtype).element_size()`
			`key_cache_shape = (num_blocks, num_heads, head_size // x, block_size, x)`
			`key_caches = []`
			`for _ in range(num_layers):`
			`key_cache = torch.empty(size=key_cache_shape,`
			`dtype=torch_dtype,`
			`device=device)`
[Minor] More fix of test_cache.py CI test failure (#2750) 2024-02-06 11:38:38 -08:00			`if cache_dtype == 'fp8_e5m2':`
Support FP8-E5M2 KV Cache (#2279) Co-authored-by: zhaoyang <zhao.yang16@zte.com.cn> Co-authored-by: Zhuohan Li <zhuohan123@gmail.com> 2024-01-29 08:43:54 +08:00			`_generate_random_fp8_e5m2(key_cache, -scale, scale)`
[Minor] More fix of test_cache.py CI test failure (#2750) 2024-02-06 11:38:38 -08:00			`elif torch_dtype in [torch.half, torch.bfloat16, torch.float]:`
			`key_cache.uniform_(-scale, scale)`
			`else:`
			`raise ValueError(`
			`f"Does not support key cache of type {cache_dtype}")`
Support FP8-E5M2 KV Cache (#2279) Co-authored-by: zhaoyang <zhao.yang16@zte.com.cn> Co-authored-by: Zhuohan Li <zhuohan123@gmail.com> 2024-01-29 08:43:54 +08:00			`key_caches.append(key_cache)`

			`value_cache_shape = (num_blocks, num_heads, head_size, block_size)`
			`value_caches = []`
			`for _ in range(num_layers):`
			`value_cache = torch.empty(size=value_cache_shape,`
			`dtype=torch_dtype,`
			`device=device)`
[Minor] More fix of test_cache.py CI test failure (#2750) 2024-02-06 11:38:38 -08:00			`if cache_dtype == 'fp8_e5m2':`
Support FP8-E5M2 KV Cache (#2279) Co-authored-by: zhaoyang <zhao.yang16@zte.com.cn> Co-authored-by: Zhuohan Li <zhuohan123@gmail.com> 2024-01-29 08:43:54 +08:00			`_generate_random_fp8_e5m2(value_cache, -scale, scale)`
[Minor] More fix of test_cache.py CI test failure (#2750) 2024-02-06 11:38:38 -08:00			`elif torch_dtype in [torch.half, torch.bfloat16, torch.float]:`
			`value_cache.uniform_(-scale, scale)`
			`else:`
			`raise ValueError(`
			`f"Does not support value cache of type {cache_dtype}")`
Support FP8-E5M2 KV Cache (#2279) Co-authored-by: zhaoyang <zhao.yang16@zte.com.cn> Co-authored-by: Zhuohan Li <zhuohan123@gmail.com> 2024-01-29 08:43:54 +08:00			`value_caches.append(value_cache)`
			`return key_caches, value_caches`
Measure model memory usage (#3120) 2024-03-07 11:42:42 -08:00

			`class measure_cuda_memory:`

			`def __init__(self, device=None):`
			`self.device = device`

			`def current_memory_usage(self) -> float:`
			`# Return the memory usage in bytes.`
			`torch.cuda.reset_peak_memory_stats(self.device)`
			`mem = torch.cuda.max_memory_allocated(self.device)`
			`return mem`

			`def __enter__(self):`
			`self.initial_memory = self.current_memory_usage()`
			`# This allows us to call methods of the context manager if needed`
			`return self`

			`def __exit__(self, exc_type, exc_val, exc_tb):`
			`self.final_memory = self.current_memory_usage()`
			`self.consumed_memory = self.final_memory - self.initial_memory`

			`# Force garbage collection`
			`gc.collect()`