Skip to content

Vllm in process

VllmInProcess(*, model, vllm_kwargs=None, lazy=False, additional_kwargs=None, **default_request_kwargs)

Bases: LLM

In-process vLLM backend using vllm.LLM.chat() so the model's chat template is applied automatically.

Supports guided decoding via StructuredOutputsParams(json=...).

In offline mode, vLLM does not automatically split reasoning vs final content for you; we do it here using the configured ReasoningParser (and a Harmony fallback).

Source code in src/kibad_llm/llms/vllm_in_process.py
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
def __init__(
    self,
    *,
    model: str,
    vllm_kwargs: dict[str, Any] | None = None,
    lazy: bool = False,
    # for compatibility with other LlamaIndex LLMs (but directly supported kwargs take precedence)
    additional_kwargs: dict[str, Any] | None = None,
    **default_request_kwargs: Any,
) -> None:
    self._model_name = model
    self._vllm_kwargs = vllm_kwargs or {}
    if not lazy:
        # trigger vLLM initialization now (instead of waiting for first call)
        # so that any errors are raised during LLM setup instead of at call time
        _ = self.llm
        _ = self.reasoning_parser

    self._default_request_kwargs: dict[str, Any] = additional_kwargs or {}
    self._default_request_kwargs.update(default_request_kwargs)

destroy()

Clean up vLLM resources.

Source code in src/kibad_llm/llms/vllm_in_process.py
123
124
125
126
127
128
129
def destroy(self) -> None:
    """Clean up vLLM resources."""
    if hasattr(self, "_llm"):
        del self._llm
    if hasattr(self, "_reasoning_parser"):
        del self._reasoning_parser
    cleanup()

get_reasoning_from_chat_response(response)

Extract reasoning from a chat response.

Source code in src/kibad_llm/llms/vllm_in_process.py
179
180
181
182
183
184
185
186
187
188
189
190
191
def get_reasoning_from_chat_response(self, response: ChatResponse) -> str | None:
    """Extract reasoning from a chat response."""

    # don't attempt extraction if no reasoning parser configured (and thus don't raise errors)
    if self.reasoning_parser is None:
        return None

    reasoning = response.message.additional_kwargs.get("reasoning")
    if not isinstance(reasoning, str):
        raise ReasoningExtractionError("Could not extract reasoning from chat response.")
    if not reasoning.strip():
        raise EmptyReasoningError("Extracted reasoning is empty.")
    return reasoning