Code
cookbook/11_models/vllm/async_tool_use.py
"""Run `uv pip install` to install dependencies."""
import asyncio
from agno.agent import Agent
from agno.models.vllm import VLLM
from agno.tools.hackernews import HackerNewsTools
agent = Agent(
model=VLLM(id="Qwen/Qwen2.5-7B-Instruct", top_k=20, enable_thinking=False),
tools=[HackerNewsTools()],
markdown=True,
)
asyncio.run(agent.aprint_response("Whats happening in France?", stream=True))
Start vLLM server
vllm serve Qwen/Qwen2.5-7B-Instruct \
--enable-auto-tool-choice \
--tool-call-parser hermes \
--dtype float16 \
--max-model-len 8192 \
--gpu-memory-utilization 0.9
Was this page helpful?