vllm/tests/entrypoints/llm/test_generate.py

import weakref
from typing import List

import pytest

from vllm import LLM, RequestOutput, SamplingParams

from ...conftest import cleanup
from ..openai.test_vision import TEST_IMAGE_URLS

MODEL_NAME = "facebook/opt-125m"

PROMPTS = [
    "Hello, my name is",
    "The president of the United States is",
    "The capital of France is",
    "The future of AI is",
]

TOKEN_IDS = [
    [0],
    [0, 1],
    [0, 2, 1],
    [0, 3, 1, 2],
]


@pytest.fixture(scope="module")
def llm():
    # pytest caches the fixture so we use weakref.proxy to
    # enable garbage collection
    llm = LLM(model=MODEL_NAME,
              max_num_batched_tokens=4096,
              tensor_parallel_size=1,
              gpu_memory_utilization=0.10,
              enforce_eager=True)

    with llm.deprecate_legacy_api():
        yield weakref.proxy(llm)

        del llm

    cleanup()


def assert_outputs_equal(o1: List[RequestOutput], o2: List[RequestOutput]):
    assert [o.outputs for o in o1] == [o.outputs for o in o2]


@pytest.mark.skip_global_cleanup
@pytest.mark.parametrize('prompt', PROMPTS)
def test_v1_v2_api_consistency_single_prompt_string(llm: LLM, prompt):
    sampling_params = SamplingParams(temperature=0.0, top_p=1.0)

    with pytest.warns(DeprecationWarning, match="'prompts'"):
        v1_output = llm.generate(prompts=prompt,
                                 sampling_params=sampling_params)

    v2_output = llm.generate(prompt, sampling_params=sampling_params)
    assert_outputs_equal(v1_output, v2_output)

    v2_output = llm.generate({"prompt": prompt},
                             sampling_params=sampling_params)
    assert_outputs_equal(v1_output, v2_output)


@pytest.mark.skip_global_cleanup
@pytest.mark.parametrize('prompt_token_ids', TOKEN_IDS)
def test_v1_v2_api_consistency_single_prompt_tokens(llm: LLM,
                                                    prompt_token_ids):
    sampling_params = SamplingParams(temperature=0.0, top_p=1.0)

    with pytest.warns(DeprecationWarning, match="'prompt_token_ids'"):
        v1_output = llm.generate(prompt_token_ids=prompt_token_ids,
                                 sampling_params=sampling_params)

    v2_output = llm.generate({"prompt_token_ids": prompt_token_ids},
                             sampling_params=sampling_params)
    assert_outputs_equal(v1_output, v2_output)


@pytest.mark.skip_global_cleanup
def test_v1_v2_api_consistency_multi_prompt_string(llm: LLM):
    sampling_params = SamplingParams(temperature=0.0, top_p=1.0)

    with pytest.warns(DeprecationWarning, match="'prompts'"):
        v1_output = llm.generate(prompts=PROMPTS,
                                 sampling_params=sampling_params)

    v2_output = llm.generate(PROMPTS, sampling_params=sampling_params)
    assert_outputs_equal(v1_output, v2_output)

    v2_output = llm.generate(
        [{
            "prompt": p
        } for p in PROMPTS],
        sampling_params=sampling_params,
    )
    assert_outputs_equal(v1_output, v2_output)


@pytest.mark.skip_global_cleanup
def test_v1_v2_api_consistency_multi_prompt_tokens(llm: LLM):
    sampling_params = SamplingParams(temperature=0.0, top_p=1.0)

    with pytest.warns(DeprecationWarning, match="'prompt_token_ids'"):
        v1_output = llm.generate(prompt_token_ids=TOKEN_IDS,
                                 sampling_params=sampling_params)

    v2_output = llm.generate(
        [{
            "prompt_token_ids": p
        } for p in TOKEN_IDS],
        sampling_params=sampling_params,
    )
    assert_outputs_equal(v1_output, v2_output)


@pytest.mark.skip_global_cleanup
def test_multiple_sampling_params(llm: LLM):
    sampling_params = [
        SamplingParams(temperature=0.01, top_p=0.95),
        SamplingParams(temperature=0.3, top_p=0.95),
        SamplingParams(temperature=0.7, top_p=0.95),
        SamplingParams(temperature=0.99, top_p=0.95),
    ]

    # Multiple SamplingParams should be matched with each prompt
    outputs = llm.generate(PROMPTS, sampling_params=sampling_params)
    assert len(PROMPTS) == len(outputs)

    # Exception raised, if the size of params does not match the size of prompts
    with pytest.raises(ValueError):
        outputs = llm.generate(PROMPTS, sampling_params=sampling_params[:3])

    # Single SamplingParams should be applied to every prompt
    single_sampling_params = SamplingParams(temperature=0.3, top_p=0.95)
    outputs = llm.generate(PROMPTS, sampling_params=single_sampling_params)
    assert len(PROMPTS) == len(outputs)

    # sampling_params is None, default params should be applied
    outputs = llm.generate(PROMPTS, sampling_params=None)
    assert len(PROMPTS) == len(outputs)


def test_chat():

    llm = LLM(model="meta-llama/Meta-Llama-3-8B-Instruct")

    prompt1 = "Explain the concept of entropy."
    messages = [
        {
            "role": "system",
            "content": "You are a helpful assistant"
        },
        {
            "role": "user",
            "content": prompt1
        },
    ]
    outputs = llm.chat(messages)
    assert len(outputs) == 1


def test_multi_chat():

    llm = LLM(model="meta-llama/Meta-Llama-3-8B-Instruct")

    prompt1 = "Explain the concept of entropy."
    prompt2 = "Explain what among us is."

    conversation1 = [
        {
            "role": "system",
            "content": "You are a helpful assistant"
        },
        {
            "role": "user",
            "content": prompt1
        },
    ]

    conversation2 = [
        {
            "role": "system",
            "content": "You are a helpful assistant"
        },
        {
            "role": "user",
            "content": prompt2
        },
    ]

    messages = [conversation1, conversation2]

    outputs = llm.chat(messages)
    assert len(outputs) == 2


@pytest.mark.parametrize("image_urls",
                         [[TEST_IMAGE_URLS[0], TEST_IMAGE_URLS[1]]])
def test_chat_multi_image(image_urls: List[str]):
    llm = LLM(
        model="microsoft/Phi-3.5-vision-instruct",
        dtype="bfloat16",
        max_model_len=4096,
        max_num_seqs=5,
        enforce_eager=True,
        trust_remote_code=True,
        limit_mm_per_prompt={"image": 2},
    )

    messages = [{
        "role":
        "user",
        "content": [
            *({
                "type": "image_url",
                "image_url": {
                    "url": image_url
                }
            } for image_url in image_urls),
            {
                "type": "text",
                "text": "What's in this image?"
            },
        ],
    }]
    outputs = llm.chat(messages)
    assert len(outputs) >= 0
[Core] Consolidate prompt arguments to LLM engines (#4328) Co-authored-by: Roger Wang <ywang@roblox.com> 2024-05-29 04:29:31 +08:00			`import weakref`
			`from typing import List`

[Frontend] multiple sampling params support (#3570) 2024-04-20 00:11:57 -07:00			`import pytest`

[Core] Consolidate prompt arguments to LLM engines (#4328) Co-authored-by: Roger Wang <ywang@roblox.com> 2024-05-29 04:29:31 +08:00			`from vllm import LLM, RequestOutput, SamplingParams`

[CI/Build] [3/3] Reorganize entrypoints tests (#5966) 2024-06-30 12:58:49 +08:00			`from ...conftest import cleanup`
[Frontend] Multimodal support in offline chat (#8098) 2024-09-04 13:22:17 +08:00			`from ..openai.test_vision import TEST_IMAGE_URLS`
[Core] Consolidate prompt arguments to LLM engines (#4328) Co-authored-by: Roger Wang <ywang@roblox.com> 2024-05-29 04:29:31 +08:00
			`MODEL_NAME = "facebook/opt-125m"`

			`PROMPTS = [`
			`"Hello, my name is",`
			`"The president of the United States is",`
			`"The capital of France is",`
			`"The future of AI is",`
			`]`
[Frontend] multiple sampling params support (#3570) 2024-04-20 00:11:57 -07:00
[Core] Consolidate prompt arguments to LLM engines (#4328) Co-authored-by: Roger Wang <ywang@roblox.com> 2024-05-29 04:29:31 +08:00			`TOKEN_IDS = [`
			`[0],`
			`[0, 1],`
			`[0, 2, 1],`
			`[0, 3, 1, 2],`
			`]`
[Frontend] multiple sampling params support (#3570) 2024-04-20 00:11:57 -07:00
[Core] Consolidate prompt arguments to LLM engines (#4328) Co-authored-by: Roger Wang <ywang@roblox.com> 2024-05-29 04:29:31 +08:00
			`@pytest.fixture(scope="module")`
			`def llm():`
			`# pytest caches the fixture so we use weakref.proxy to`
			`# enable garbage collection`
			`llm = LLM(model=MODEL_NAME,`
[Frontend] multiple sampling params support (#3570) 2024-04-20 00:11:57 -07:00			`max_num_batched_tokens=4096,`
[Core] Consolidate prompt arguments to LLM engines (#4328) Co-authored-by: Roger Wang <ywang@roblox.com> 2024-05-29 04:29:31 +08:00			`tensor_parallel_size=1,`
			`gpu_memory_utilization=0.10,`
			`enforce_eager=True)`

			`with llm.deprecate_legacy_api():`
			`yield weakref.proxy(llm)`

			`del llm`

			`cleanup()`


			`def assert_outputs_equal(o1: List[RequestOutput], o2: List[RequestOutput]):`
			`assert [o.outputs for o in o1] == [o.outputs for o in o2]`


Revert "rename PromptInputs and inputs with backward compatibility (#8760) (#8810) 2024-09-25 10:36:26 -07:00			`@pytest.mark.skip_global_cleanup`
			`@pytest.mark.parametrize('prompt', PROMPTS)`
			`def test_v1_v2_api_consistency_single_prompt_string(llm: LLM, prompt):`
			`sampling_params = SamplingParams(temperature=0.0, top_p=1.0)`

			`with pytest.warns(DeprecationWarning, match="'prompts'"):`
			`v1_output = llm.generate(prompts=prompt,`
			`sampling_params=sampling_params)`

			`v2_output = llm.generate(prompt, sampling_params=sampling_params)`
			`assert_outputs_equal(v1_output, v2_output)`

			`v2_output = llm.generate({"prompt": prompt},`
			`sampling_params=sampling_params)`
			`assert_outputs_equal(v1_output, v2_output)`


[Core] Consolidate prompt arguments to LLM engines (#4328) Co-authored-by: Roger Wang <ywang@roblox.com> 2024-05-29 04:29:31 +08:00			`@pytest.mark.skip_global_cleanup`
			`@pytest.mark.parametrize('prompt_token_ids', TOKEN_IDS)`
			`def test_v1_v2_api_consistency_single_prompt_tokens(llm: LLM,`
			`prompt_token_ids):`
			`sampling_params = SamplingParams(temperature=0.0, top_p=1.0)`

			`with pytest.warns(DeprecationWarning, match="'prompt_token_ids'"):`
			`v1_output = llm.generate(prompt_token_ids=prompt_token_ids,`
			`sampling_params=sampling_params)`

			`v2_output = llm.generate({"prompt_token_ids": prompt_token_ids},`
			`sampling_params=sampling_params)`
			`assert_outputs_equal(v1_output, v2_output)`


Revert "rename PromptInputs and inputs with backward compatibility (#8760) (#8810) 2024-09-25 10:36:26 -07:00			`@pytest.mark.skip_global_cleanup`
			`def test_v1_v2_api_consistency_multi_prompt_string(llm: LLM):`
			`sampling_params = SamplingParams(temperature=0.0, top_p=1.0)`

			`with pytest.warns(DeprecationWarning, match="'prompts'"):`
			`v1_output = llm.generate(prompts=PROMPTS,`
			`sampling_params=sampling_params)`

			`v2_output = llm.generate(PROMPTS, sampling_params=sampling_params)`
			`assert_outputs_equal(v1_output, v2_output)`

			`v2_output = llm.generate(`
			`[{`
			`"prompt": p`
			`} for p in PROMPTS],`
			`sampling_params=sampling_params,`
			`)`
			`assert_outputs_equal(v1_output, v2_output)`


[Core] Consolidate prompt arguments to LLM engines (#4328) Co-authored-by: Roger Wang <ywang@roblox.com> 2024-05-29 04:29:31 +08:00			`@pytest.mark.skip_global_cleanup`
			`def test_v1_v2_api_consistency_multi_prompt_tokens(llm: LLM):`
			`sampling_params = SamplingParams(temperature=0.0, top_p=1.0)`

			`with pytest.warns(DeprecationWarning, match="'prompt_token_ids'"):`
			`v1_output = llm.generate(prompt_token_ids=TOKEN_IDS,`
			`sampling_params=sampling_params)`

			`v2_output = llm.generate(`
			`[{`
			`"prompt_token_ids": p`
			`} for p in TOKEN_IDS],`
			`sampling_params=sampling_params,`
			`)`
			`assert_outputs_equal(v1_output, v2_output)`
[Frontend] multiple sampling params support (#3570) 2024-04-20 00:11:57 -07:00

[Core] Consolidate prompt arguments to LLM engines (#4328) Co-authored-by: Roger Wang <ywang@roblox.com> 2024-05-29 04:29:31 +08:00			`@pytest.mark.skip_global_cleanup`
			`def test_multiple_sampling_params(llm: LLM):`
[Frontend] multiple sampling params support (#3570) 2024-04-20 00:11:57 -07:00			`sampling_params = [`
			`SamplingParams(temperature=0.01, top_p=0.95),`
			`SamplingParams(temperature=0.3, top_p=0.95),`
			`SamplingParams(temperature=0.7, top_p=0.95),`
			`SamplingParams(temperature=0.99, top_p=0.95),`
			`]`

			`# Multiple SamplingParams should be matched with each prompt`
[Core] Consolidate prompt arguments to LLM engines (#4328) Co-authored-by: Roger Wang <ywang@roblox.com> 2024-05-29 04:29:31 +08:00			`outputs = llm.generate(PROMPTS, sampling_params=sampling_params)`
			`assert len(PROMPTS) == len(outputs)`
[Frontend] multiple sampling params support (#3570) 2024-04-20 00:11:57 -07:00
			`# Exception raised, if the size of params does not match the size of prompts`
			`with pytest.raises(ValueError):`
[Core] Consolidate prompt arguments to LLM engines (#4328) Co-authored-by: Roger Wang <ywang@roblox.com> 2024-05-29 04:29:31 +08:00			`outputs = llm.generate(PROMPTS, sampling_params=sampling_params[:3])`
[Frontend] multiple sampling params support (#3570) 2024-04-20 00:11:57 -07:00
			`# Single SamplingParams should be applied to every prompt`
			`single_sampling_params = SamplingParams(temperature=0.3, top_p=0.95)`
[Core] Consolidate prompt arguments to LLM engines (#4328) Co-authored-by: Roger Wang <ywang@roblox.com> 2024-05-29 04:29:31 +08:00			`outputs = llm.generate(PROMPTS, sampling_params=single_sampling_params)`
			`assert len(PROMPTS) == len(outputs)`
[Frontend] multiple sampling params support (#3570) 2024-04-20 00:11:57 -07:00
			`# sampling_params is None, default params should be applied`
[Core] Consolidate prompt arguments to LLM engines (#4328) Co-authored-by: Roger Wang <ywang@roblox.com> 2024-05-29 04:29:31 +08:00			`outputs = llm.generate(PROMPTS, sampling_params=None)`
			`assert len(PROMPTS) == len(outputs)`
Chat method for offline llm (#5049) Co-authored-by: nunjunj <ray@g-3ff9f30f2ed650001.c.vllm-405802.internal> Co-authored-by: nunjunj <ray@g-1df6075697c3f0001.c.vllm-405802.internal> Co-authored-by: nunjunj <ray@g-c5a2c23abc49e0001.c.vllm-405802.internal> Co-authored-by: Cyrus Leung <cyrus.tl.leung@gmail.com> Co-authored-by: DarkLight1337 <tlleungac@connect.ust.hk> 2024-08-16 09:41:34 +07:00

			`def test_chat():`

			`llm = LLM(model="meta-llama/Meta-Llama-3-8B-Instruct")`

			`prompt1 = "Explain the concept of entropy."`
			`messages = [`
			`{`
			`"role": "system",`
			`"content": "You are a helpful assistant"`
			`},`
			`{`
			`"role": "user",`
			`"content": prompt1`
			`},`
			`]`
			`outputs = llm.chat(messages)`
			`assert len(outputs) == 1`
[Frontend] Multimodal support in offline chat (#8098) 2024-09-04 13:22:17 +08:00

[Frontend] Batch inference for llm.chat() API (#8648) Co-authored-by: Cyrus Leung <cyrus.tl.leung@gmail.com> Co-authored-by: Cyrus Leung <tlleungac@connect.ust.hk> Co-authored-by: Roger Wang <ywang@roblox.com> Co-authored-by: Roger Wang <136131678+ywang96@users.noreply.github.com> 2024-09-24 12:44:11 -04:00			`def test_multi_chat():`

			`llm = LLM(model="meta-llama/Meta-Llama-3-8B-Instruct")`

			`prompt1 = "Explain the concept of entropy."`
			`prompt2 = "Explain what among us is."`

			`conversation1 = [`
			`{`
			`"role": "system",`
			`"content": "You are a helpful assistant"`
			`},`
			`{`
			`"role": "user",`
			`"content": prompt1`
			`},`
			`]`

			`conversation2 = [`
			`{`
			`"role": "system",`
			`"content": "You are a helpful assistant"`
			`},`
			`{`
			`"role": "user",`
			`"content": prompt2`
			`},`
			`]`

			`messages = [conversation1, conversation2]`

			`outputs = llm.chat(messages)`
			`assert len(outputs) == 2`


[Frontend] Multimodal support in offline chat (#8098) 2024-09-04 13:22:17 +08:00			`@pytest.mark.parametrize("image_urls",`
			`[[TEST_IMAGE_URLS[0], TEST_IMAGE_URLS[1]]])`
			`def test_chat_multi_image(image_urls: List[str]):`
			`llm = LLM(`
			`model="microsoft/Phi-3.5-vision-instruct",`
			`dtype="bfloat16",`
			`max_model_len=4096,`
			`max_num_seqs=5,`
			`enforce_eager=True,`
			`trust_remote_code=True,`
			`limit_mm_per_prompt={"image": 2},`
			`)`

			`messages = [{`
			`"role":`
			`"user",`
			`"content": [`
			`*({`
			`"type": "image_url",`
			`"image_url": {`
			`"url": image_url`
			`}`
			`} for image_url in image_urls),`
			`{`
			`"type": "text",`
			`"text": "What's in this image?"`
			`},`
			`],`
			`}]`
			`outputs = llm.chat(messages)`
			`assert len(outputs) >= 0`