cleanup

ParthSareen · ParthSareen · commit 3a67d2740542 · 2025-09-23T18:12:39.000-07:00
diff --git a/examples/README.md b/examples/README.md
@@ -43,15 +43,30 @@ See [ollama/docs/api.md](https://github.com/ollama/ollama/blob/main/docs/api.md)
 
 `OLLAMA_API_KEY` is required. You can get one from [ollama.com/settings/keys](https://ollama.com/settings/keys).
 
-- [web-search-fetch.py](web-search-fetch.py)
+- [web-search.py](web-search.py)
 
 #### MCP server
 
 ```sh
-uv run examples/mcp-web-search-and-fetch.py
+uv run examples/web-search-mcp.py
 ```
 
-- [mcp_web_search_crawl_server.py](mcp_web_search_crawl_server.py)
+Configuration to use with an MCP client:
+
+```json
+{
+  "mcpServers": {
+    "web_search": {
+      "type": "stdio",
+      "command": "uv",
+      "args": ["run", "path/to/ollama-python/examples/web-search-mcp.py"],
+      "env": { "OLLAMA_API_KEY": "api-key" }
+    }
+  }
+}
+```
+
+- [web-search-mcp.py](web-search-mcp.py)
 
 ### Multimodal with Images - Chat with a multimodal (image chat) model
 
diff --git a/examples/web-search-mcp.py b/examples/web-search-mcp.py
@@ -0,0 +1,116 @@
+# /// script
+# requires-python = ">=3.11"
+# dependencies = [
+#   "mcp",
+#   "rich",
+#   "ollama",
+# ]
+# ///
+"""
+MCP stdio server exposing Ollama web_search and web_fetch as tools.
+
+Environment:
+- OLLAMA_API_KEY (required): if set, will be used as Authorization header.
+"""
+
+from __future__ import annotations
+
+import asyncio
+from typing import Any, Dict
+
+from ollama import Client
+
+try:
+  # Preferred high-level API (if available)
+  from mcp.server.fastmcp import FastMCP  # type: ignore
+
+  _FASTMCP_AVAILABLE = True
+except Exception:
+  _FASTMCP_AVAILABLE = False
+
+if not _FASTMCP_AVAILABLE:
+  # Fallback to the low-level stdio server API
+  from mcp.server import Server  # type: ignore
+  from mcp.server.stdio import stdio_server  # type: ignore
+
+
+client = Client()
+
+
+def _web_search_impl(query: str, max_results: int = 3) -> Dict[str, Any]:
+  res = client.web_search(query=query, max_results=max_results)
+  return res.model_dump()
+
+
+def _web_fetch_impl(url: str) -> Dict[str, Any]:
+  res = client.web_fetch(url=url)
+  return res.model_dump()
+
+
+if _FASTMCP_AVAILABLE:
+  app = FastMCP('ollama-search-fetch')
+
+  @app.tool()
+  def web_search(query: str, max_results: int = 3) -> Dict[str, Any]:
+    """
+    Perform a web search using Ollama's hosted search API.
+
+    Args:
+      query: The search query to run.
+      max_results: Maximum results to return (default: 3).
+
+    Returns:
+      JSON-serializable dict matching ollama.WebSearchResponse.model_dump()
+    """
+
+    return _web_search_impl(query=query, max_results=max_results)
+
+  @app.tool()
+  def web_fetch(url: str) -> Dict[str, Any]:
+    """
+    Fetch the content of a web page for the provided URL.
+
+    Args:
+      url: The absolute URL to fetch.
+
+    Returns:
+      JSON-serializable dict matching ollama.WebFetchResponse.model_dump()
+    """
+
+    return _web_fetch_impl(url=url)
+
+  if __name__ == '__main__':
+    app.run()
+
+else:
+  server = Server('ollama-search-fetch')  # type: ignore[name-defined]
+
+  @server.tool()  # type: ignore[attr-defined]
+  async def web_search(query: str, max_results: int = 3) -> Dict[str, Any]:
+    """
+    Perform a web search using Ollama's hosted search API.
+
+    Args:
+      query: The search query to run.
+      max_results: Maximum results to return (default: 3).
+    """
+
+    return await asyncio.to_thread(_web_search_impl, query, max_results)
+
+  @server.tool()  # type: ignore[attr-defined]
+  async def web_fetch(url: str) -> Dict[str, Any]:
+    """
+    Fetch the content of a web page for the provided URL.
+
+    Args:
+      url: The absolute URL to fetch.
+    """
+
+    return await asyncio.to_thread(_web_fetch_impl, url)
+
+  async def _main() -> None:
+    async with stdio_server() as (read, write):  # type: ignore[name-defined]
+      await server.run(read, write)  # type: ignore[attr-defined]
+
+  if __name__ == '__main__':
+    asyncio.run(_main())
diff --git a/examples/web-search.py b/examples/web-search.py
@@ -0,0 +1,85 @@
+# /// script
+# requires-python = ">=3.11"
+# dependencies = [
+#     "rich",
+#     "ollama",
+# ]
+# ///
+from typing import Union
+
+from rich import print
+
+from ollama import WebFetchResponse, WebSearchResponse, chat, web_fetch, web_search
+
+
+def format_tool_results(
+  results: Union[WebSearchResponse, WebFetchResponse],
+  user_search: str,
+):
+  output = []
+  if isinstance(results, WebSearchResponse):
+    output.append(f'Search results for "{user_search}":')
+    for result in results.results:
+      output.append(f'{result.title}' if result.title else f'{result.content}')
+      output.append(f'   URL: {result.url}')
+      output.append(f'   Content: {result.content}')
+      output.append('')
+    return '\n'.join(output).rstrip()
+
+  elif isinstance(results, WebFetchResponse):
+    output.append(f'Fetch results for "{user_search}":')
+    output.extend(
+      [
+        f'Title: {results.title}',
+        f'URL: {user_search}' if user_search else '',
+        f'Content: {results.content}',
+      ]
+    )
+    if results.links:
+      output.append(f'Links: {", ".join(results.links)}')
+    output.append('')
+    return '\n'.join(output).rstrip()
+
+
+# client = Client(headers={'Authorization': f"Bearer {os.getenv('OLLAMA_API_KEY')}"} if api_key else None)
+available_tools = {'web_search': web_search, 'web_fetch': web_fetch}
+
+query = "what is ollama's new engine"
+print('Query: ', query)
+
+messages = [{'role': 'user', 'content': query}]
+while True:
+  response = chat(model='qwen3', messages=messages, tools=[web_search, web_fetch], think=True)
+  if response.message.thinking:
+    print('Thinking: ')
+    print(response.message.thinking + '\n\n')
+  if response.message.content:
+    print('Content: ')
+    print(response.message.content + '\n')
+
+  messages.append(response.message)
+
+  if response.message.tool_calls:
+    for tool_call in response.message.tool_calls:
+      function_to_call = available_tools.get(tool_call.function.name)
+      if function_to_call:
+        args = tool_call.function.arguments
+        result: Union[WebSearchResponse, WebFetchResponse] = function_to_call(**args)
+        print('Result from tool call name:', tool_call.function.name, 'with arguments:')
+        print(args)
+        print()
+
+        user_search = args.get('query', '') or args.get('url', '')
+        formatted_tool_results = format_tool_results(result, user_search=user_search)
+
+        print(formatted_tool_results[:300])
+        print()
+
+        # caps the result at ~2000 tokens
+        messages.append({'role': 'tool', 'content': formatted_tool_results[: 2000 * 4], 'tool_name': tool_call.function.name})
+      else:
+        print(f'Tool {tool_call.function.name} not found')
+        messages.append({'role': 'tool', 'content': f'Tool {tool_call.function.name} not found', 'tool_name': tool_call.function.name})
+  else:
+    # no more tool calls, we can stop the loop
+    break