Skip to content

ant_ai.llm.protocol

ChatLLM

Bases: Protocol

Interface for a language model that generates chat responses.

Every backend is constructed as Backend(model, *, api_key=None, api_base=None) and exposes the three as attributes. A caller can therefore point any backend at its own deployment (a vLLM server, a proxy, …) without knowing which one it holds, and keep the secret under its own name instead of the one the provider's SDK reads from the environment.

Attributes:

Name Type Description
model str

Model identifier in the backend's own naming scheme.

api_key str | None

Credential for the endpoint, or None to let the backend fall back to its provider's environment variable.

api_base str | None

Endpoint URL, or None for the provider's default.

Source code in src/ant_ai/llm/protocol.py
 11
 12
 13
 14
 15
 16
 17
 18
 19
 20
 21
 22
 23
 24
 25
 26
 27
 28
 29
 30
 31
 32
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
class ChatLLM(Protocol):
    """Interface for a language model that generates chat responses.

    Every backend is constructed as `Backend(model, *, api_key=None,
    api_base=None)` and exposes the three as attributes. A caller can therefore
    point any backend at its own deployment (a vLLM server, a proxy, …) without
    knowing which one it holds, and keep the secret under its own name instead
    of the one the provider's SDK reads from the environment.

    Attributes:
        model: Model identifier in the backend's own naming scheme.
        api_key: Credential for the endpoint, or None to let the backend fall
            back to its provider's environment variable.
        api_base: Endpoint URL, or None for the provider's default.
    """

    model: str
    api_key: str | None
    api_base: str | None

    def invoke(
        self,
        messages: list[Message],
        *,
        ctx: InvocationContext | None = None,
        tools: list | None = None,
        response_format: dict | type[BaseModel] | None = None,
    ) -> ChatLLMResponse:
        """Send messages and return a complete response synchronously.

        Args:
            messages: Conversation history to send to the model.
            ctx: Invocation context, or None if not available.
            tools: Tool schemas to expose to the model, or None for no tools.
            response_format: Constrain the output to a JSON schema or Pydantic model.

        Returns:
            The complete model response.
        """
        raise NotImplementedError

    async def ainvoke(
        self,
        messages: list[Message],
        *,
        ctx: InvocationContext | None = None,
        tools: list | None = None,
        response_format: dict | type[BaseModel] | None = None,
    ) -> ChatLLMResponse:
        """Send messages and return a complete response asynchronously.

        Args:
            messages: Conversation history to send to the model.
            ctx: Invocation context, or None if not available.
            tools: Tool schemas to expose to the model, or None for no tools.
            response_format: Constrain the output to a JSON schema or Pydantic model.

        Returns:
            The complete model response.
        """
        raise NotImplementedError

    def stream(
        self,
        messages: list[Message],
        *,
        ctx: InvocationContext | None = None,
        tools: list | None = None,
        response_format: dict | type[BaseModel] | None = None,
    ) -> AsyncIterator[ChatLLMStreamChunk]:
        """Send messages and stream the response as chunks.

        Backends that generate tokens incrementally (e.g. OpenAI, LiteLLM)
        override this for true token-by-token delivery. The default here
        falls back to `ainvoke()` and re-emits its result as a single chunk,
        so any `ChatLLM` implementation — including test doubles and
        backends with no incremental API — supports `.stream()` without
        extra work, just without the live granularity.

        Args:
            messages: Conversation history to send to the model.
            ctx: Invocation context, or None if not available.
            tools: Tool schemas to expose to the model, or None for no tools.
            response_format: Constrain the output to a JSON schema or Pydantic model.

        Returns:
            An async iterator of response chunks.
        """

        async def gen() -> AsyncIterator[ChatLLMStreamChunk]:
            response: ChatLLMResponse = await self.ainvoke(
                messages, ctx=ctx, tools=tools, response_format=response_format
            )
            reasoning = getattr(response, "reasoning", None)
            if response.message.content or reasoning:
                yield ChatLLMStreamChunk(
                    delta=MessageChunk(
                        role=response.message.role,
                        delta=response.message.content or "",
                    ),
                    reasoning_delta=reasoning,
                )
            for i, tool_call in enumerate(response.tool_calls or []):
                yield ChatLLMStreamChunk(
                    delta=MessageChunk(role="assistant", delta=""),
                    tool_calls={
                        "index": i,
                        "id": tool_call.id,
                        "name": tool_call.function.name,
                        "arguments": tool_call.function.arguments,
                    },
                )

        return gen()

invoke

invoke(
    messages: list[Message],
    *,
    ctx: InvocationContext | None = None,
    tools: list | None = None,
    response_format: dict | type[BaseModel] | None = None,
) -> ChatLLMResponse

Send messages and return a complete response synchronously.

Parameters:

Name Type Description Default
messages list[Message]

Conversation history to send to the model.

required
ctx InvocationContext | None

Invocation context, or None if not available.

None
tools list | None

Tool schemas to expose to the model, or None for no tools.

None
response_format dict | type[BaseModel] | None

Constrain the output to a JSON schema or Pydantic model.

None

Returns:

Type Description
ChatLLMResponse

The complete model response.

Source code in src/ant_ai/llm/protocol.py
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
def invoke(
    self,
    messages: list[Message],
    *,
    ctx: InvocationContext | None = None,
    tools: list | None = None,
    response_format: dict | type[BaseModel] | None = None,
) -> ChatLLMResponse:
    """Send messages and return a complete response synchronously.

    Args:
        messages: Conversation history to send to the model.
        ctx: Invocation context, or None if not available.
        tools: Tool schemas to expose to the model, or None for no tools.
        response_format: Constrain the output to a JSON schema or Pydantic model.

    Returns:
        The complete model response.
    """
    raise NotImplementedError

ainvoke async

ainvoke(
    messages: list[Message],
    *,
    ctx: InvocationContext | None = None,
    tools: list | None = None,
    response_format: dict | type[BaseModel] | None = None,
) -> ChatLLMResponse

Send messages and return a complete response asynchronously.

Parameters:

Name Type Description Default
messages list[Message]

Conversation history to send to the model.

required
ctx InvocationContext | None

Invocation context, or None if not available.

None
tools list | None

Tool schemas to expose to the model, or None for no tools.

None
response_format dict | type[BaseModel] | None

Constrain the output to a JSON schema or Pydantic model.

None

Returns:

Type Description
ChatLLMResponse

The complete model response.

Source code in src/ant_ai/llm/protocol.py
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
async def ainvoke(
    self,
    messages: list[Message],
    *,
    ctx: InvocationContext | None = None,
    tools: list | None = None,
    response_format: dict | type[BaseModel] | None = None,
) -> ChatLLMResponse:
    """Send messages and return a complete response asynchronously.

    Args:
        messages: Conversation history to send to the model.
        ctx: Invocation context, or None if not available.
        tools: Tool schemas to expose to the model, or None for no tools.
        response_format: Constrain the output to a JSON schema or Pydantic model.

    Returns:
        The complete model response.
    """
    raise NotImplementedError

stream

stream(
    messages: list[Message],
    *,
    ctx: InvocationContext | None = None,
    tools: list | None = None,
    response_format: dict | type[BaseModel] | None = None,
) -> AsyncIterator[ChatLLMStreamChunk]

Send messages and stream the response as chunks.

Backends that generate tokens incrementally (e.g. OpenAI, LiteLLM) override this for true token-by-token delivery. The default here falls back to ainvoke() and re-emits its result as a single chunk, so any ChatLLM implementation — including test doubles and backends with no incremental API — supports .stream() without extra work, just without the live granularity.

Parameters:

Name Type Description Default
messages list[Message]

Conversation history to send to the model.

required
ctx InvocationContext | None

Invocation context, or None if not available.

None
tools list | None

Tool schemas to expose to the model, or None for no tools.

None
response_format dict | type[BaseModel] | None

Constrain the output to a JSON schema or Pydantic model.

None

Returns:

Type Description
AsyncIterator[ChatLLMStreamChunk]

An async iterator of response chunks.

Source code in src/ant_ai/llm/protocol.py
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
def stream(
    self,
    messages: list[Message],
    *,
    ctx: InvocationContext | None = None,
    tools: list | None = None,
    response_format: dict | type[BaseModel] | None = None,
) -> AsyncIterator[ChatLLMStreamChunk]:
    """Send messages and stream the response as chunks.

    Backends that generate tokens incrementally (e.g. OpenAI, LiteLLM)
    override this for true token-by-token delivery. The default here
    falls back to `ainvoke()` and re-emits its result as a single chunk,
    so any `ChatLLM` implementation — including test doubles and
    backends with no incremental API — supports `.stream()` without
    extra work, just without the live granularity.

    Args:
        messages: Conversation history to send to the model.
        ctx: Invocation context, or None if not available.
        tools: Tool schemas to expose to the model, or None for no tools.
        response_format: Constrain the output to a JSON schema or Pydantic model.

    Returns:
        An async iterator of response chunks.
    """

    async def gen() -> AsyncIterator[ChatLLMStreamChunk]:
        response: ChatLLMResponse = await self.ainvoke(
            messages, ctx=ctx, tools=tools, response_format=response_format
        )
        reasoning = getattr(response, "reasoning", None)
        if response.message.content or reasoning:
            yield ChatLLMStreamChunk(
                delta=MessageChunk(
                    role=response.message.role,
                    delta=response.message.content or "",
                ),
                reasoning_delta=reasoning,
            )
        for i, tool_call in enumerate(response.tool_calls or []):
            yield ChatLLMStreamChunk(
                delta=MessageChunk(role="assistant", delta=""),
                tool_calls={
                    "index": i,
                    "id": tool_call.id,
                    "name": tool_call.function.name,
                    "arguments": tool_call.function.arguments,
                },
            )

    return gen()