././@PaxHeader0000000000000000000000000000003400000000000010212 xustar0028 mtime=1785880938.9594848 llm_anthropic-0.26/0000755000175100017510000000000015234460553013751 5ustar00runnerrunner././@PaxHeader0000000000000000000000000000002600000000000010213 xustar0022 mtime=1785880929.0 llm_anthropic-0.26/LICENSE0000644000175100017510000002613515234460541014762 0ustar00runnerrunner Apache License Version 2.0, January 2004 http://www.apache.org/licenses/ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION 1. Definitions. "License" shall mean the terms and conditions for use, reproduction, and distribution as defined by Sections 1 through 9 of this document. "Licensor" shall mean the copyright owner or entity authorized by the copyright owner that is granting the License. "Legal Entity" shall mean the union of the acting entity and all other entities that control, are controlled by, or are under common control with that entity. For the purposes of this definition, "control" means (i) the power, direct or indirect, to cause the direction or management of such entity, whether by contract or otherwise, or (ii) ownership of fifty percent (50%) or more of the outstanding shares, or (iii) beneficial ownership of such entity. "You" (or "Your") shall mean an individual or Legal Entity exercising permissions granted by this License. "Source" form shall mean the preferred form for making modifications, including but not limited to software source code, documentation source, and configuration files. "Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types. "Work" shall mean the work of authorship, whether in Source or Object form, made available under the License, as indicated by a copyright notice that is included in or attached to the work (an example is provided in the Appendix below). "Derivative Works" shall mean any work, whether in Source or Object form, that is based on (or derived from) the Work and for which the editorial revisions, annotations, elaborations, or other modifications represent, as a whole, an original work of authorship. For the purposes of this License, Derivative Works shall not include works that remain separable from, or merely link (or bind by name) to the interfaces of, the Work and Derivative Works thereof. "Contribution" shall mean any work of authorship, including the original version of the Work and any modifications or additions to that Work or Derivative Works thereof, that is intentionally submitted to Licensor for inclusion in the Work by the copyright owner or by an individual or Legal Entity authorized to submit on behalf of the copyright owner. For the purposes of this definition, "submitted" means any form of electronic, verbal, or written communication sent to the Licensor or its representatives, including but not limited to communication on electronic mailing lists, source code control systems, and issue tracking systems that are managed by, or on behalf of, the Licensor for the purpose of discussing and improving the Work, but excluding communication that is conspicuously marked or otherwise designated in writing by the copyright owner as "Not a Contribution." "Contributor" shall mean Licensor and any individual or Legal Entity on behalf of whom a Contribution has been received by Licensor and subsequently incorporated within the Work. 2. Grant of Copyright License. Subject to the terms and conditions of this License, each Contributor hereby grants to You a perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable copyright license to reproduce, prepare Derivative Works of, publicly display, publicly perform, sublicense, and distribute the Work and such Derivative Works in Source or Object form. 3. Grant of Patent License. Subject to the terms and conditions of this License, each Contributor hereby grants to You a perpetual, worldwide, non-exclusive, no-charge, royalty-free, irrevocable (except as stated in this section) patent license to make, have made, use, offer to sell, sell, import, and otherwise transfer the Work, where such license applies only to those patent claims licensable by such Contributor that are necessarily infringed by their Contribution(s) alone or by combination of their Contribution(s) with the Work to which such Contribution(s) was submitted. If You institute patent litigation against any entity (including a cross-claim or counterclaim in a lawsuit) alleging that the Work or a Contribution incorporated within the Work constitutes direct or contributory patent infringement, then any patent licenses granted to You under this License for that Work shall terminate as of the date such litigation is filed. 4. Redistribution. You may reproduce and distribute copies of the Work or Derivative Works thereof in any medium, with or without modifications, and in Source or Object form, provided that You meet the following conditions: (a) You must give any other recipients of the Work or Derivative Works a copy of this License; and (b) You must cause any modified files to carry prominent notices stating that You changed the files; and (c) You must retain, in the Source form of any Derivative Works that You distribute, all copyright, patent, trademark, and attribution notices from the Source form of the Work, excluding those notices that do not pertain to any part of the Derivative Works; and (d) If the Work includes a "NOTICE" text file as part of its distribution, then any Derivative Works that You distribute must include a readable copy of the attribution notices contained within such NOTICE file, excluding those notices that do not pertain to any part of the Derivative Works, in at least one of the following places: within a NOTICE text file distributed as part of the Derivative Works; within the Source form or documentation, if provided along with the Derivative Works; or, within a display generated by the Derivative Works, if and wherever such third-party notices normally appear. The contents of the NOTICE file are for informational purposes only and do not modify the License. You may add Your own attribution notices within Derivative Works that You distribute, alongside or as an addendum to the NOTICE text from the Work, provided that such additional attribution notices cannot be construed as modifying the License. You may add Your own copyright statement to Your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of Your modifications, or for any such Derivative Works as a whole, provided Your use, reproduction, and distribution of the Work otherwise complies with the conditions stated in this License. 5. Submission of Contributions. Unless You explicitly state otherwise, any Contribution intentionally submitted for inclusion in the Work by You to the Licensor shall be under the terms and conditions of this License, without any additional terms or conditions. Notwithstanding the above, nothing herein shall supersede or modify the terms of any separate license agreement you may have executed with Licensor regarding such Contributions. 6. Trademarks. This License does not grant permission to use the trade names, trademarks, service marks, or product names of the Licensor, except as required for reasonable and customary use in describing the origin of the Work and reproducing the content of the NOTICE file. 7. Disclaimer of Warranty. Unless required by applicable law or agreed to in writing, Licensor provides the Work (and each Contributor provides its Contributions) on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied, including, without limitation, any warranties or conditions of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A PARTICULAR PURPOSE. You are solely responsible for determining the appropriateness of using or redistributing the Work and assume any risks associated with Your exercise of permissions under this License. 8. Limitation of Liability. In no event and under no legal theory, whether in tort (including negligence), contract, or otherwise, unless required by applicable law (such as deliberate and grossly negligent acts) or agreed to in writing, shall any Contributor be liable to You for damages, including any direct, indirect, special, incidental, or consequential damages of any character arising as a result of this License or out of the use or inability to use the Work (including but not limited to damages for loss of goodwill, work stoppage, computer failure or malfunction, or any and all other commercial damages or losses), even if such Contributor has been advised of the possibility of such damages. 9. Accepting Warranty or Additional Liability. While redistributing the Work or Derivative Works thereof, You may choose to offer, and charge a fee for, acceptance of support, warranty, indemnity, or other liability obligations and/or rights consistent with this License. However, in accepting such obligations, You may act only on Your own behalf and on Your sole responsibility, not on behalf of any other Contributor, and only if You agree to indemnify, defend, and hold each Contributor harmless for any liability incurred by, or claims asserted against, such Contributor by reason of your accepting any such warranty or additional liability. END OF TERMS AND CONDITIONS APPENDIX: How to apply the Apache License to your work. To apply the Apache License to your work, attach the following boilerplate notice, with the fields enclosed by brackets "[]" replaced with your own identifying information. (Don't include the brackets!) The text should be enclosed in the appropriate comment syntax for the file format. We also recommend that a file or class name and description of purpose be included on the same "printed page" as the copyright notice for easier identification within third-party archives. Copyright [yyyy] [name of copyright owner] Licensed under the Apache License, Version 2.0 (the "License"); you may not use this file except in compliance with the License. You may obtain a copy of the License at http://www.apache.org/licenses/LICENSE-2.0 Unless required by applicable law or agreed to in writing, software distributed under the License is distributed on an "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. See the License for the specific language governing permissions and limitations under the License. ././@PaxHeader0000000000000000000000000000003400000000000010212 xustar0028 mtime=1785880938.9594653 llm_anthropic-0.26/PKG-INFO0000644000175100017510000004043415234460553015053 0ustar00runnerrunnerMetadata-Version: 2.4 Name: llm-anthropic Version: 0.26 Summary: LLM access to models by Anthropic, including the Claude series Author: Simon Willison License-Expression: Apache-2.0 Project-URL: Homepage, https://github.com/simonw/llm-anthropic Project-URL: Changelog, https://github.com/simonw/llm-anthropic/releases Project-URL: Issues, https://github.com/simonw/llm-anthropic/issues Project-URL: CI, https://github.com/simonw/llm-anthropic/actions Requires-Python: >=3.10 Description-Content-Type: text/markdown License-File: LICENSE Requires-Dist: llm>=0.32 Requires-Dist: anthropic>=0.96.0 Dynamic: license-file # llm-anthropic [![PyPI](https://img.shields.io/pypi/v/llm-anthropic.svg)](https://pypi.org/project/llm-anthropic/) [![Changelog](https://img.shields.io/github/v/release/simonw/llm-anthropic?include_prereleases&label=changelog)](https://github.com/simonw/llm-anthropic/releases) [![Tests](https://github.com/simonw/llm-anthropic/actions/workflows/test.yml/badge.svg)](https://github.com/simonw/llm-anthropic/actions/workflows/test.yml) [![License](https://img.shields.io/badge/license-Apache%202.0-blue.svg)](https://github.com/simonw/llm-anthropic/blob/main/LICENSE) LLM access to models by Anthropic, including the Claude series ## Installation Install this plugin in the same environment as [LLM](https://llm.datasette.io/). ```bash llm install llm-anthropic ```
Instructions for users who need to upgrade from llm-claude-3
If you previously used `llm-claude-3` you can upgrade like this: ```bash llm install -U llm-claude-3 llm keys set anthropic --value "$(llm keys get claude)" ``` The first line will remove the previous `llm-claude-3` version and install this one, because the latest `llm-claude-3` depends on `llm-anthropic`. The second line sets the `anthropic` key to whatever value you previously used for the `claude` key.
## Usage First, set [an API key](https://console.anthropic.com/settings/keys) for Anthropic: ```bash llm keys set anthropic # Paste key here ``` You can also set the key in the environment variable `ANTHROPIC_API_KEY` Run `llm models` to list the models, and `llm models --options` to include a list of their options. Run prompts like this: ```bash llm -m claude-opus-5 'Fun facts about walruses' llm -m claude-sonnet-5 'Fun facts about pelicans' llm -m claude-haiku-4.5 'Fun facts about cormorants' ``` Image attachments are supported too: ```bash llm -m claude-sonnet-5 'describe this image' -a https://static.simonwillison.net/static/2024/pelicans.jpg llm -m claude-haiku-4.5 'extract text' -a page.png ``` The Claude 3.5 and 4 models can handle PDF files: ```bash llm -m claude-sonnet-5 'extract text' -a page.pdf ``` Anthropic's models support [schemas](https://llm.datasette.io/en/stable/schemas.html). Here's how to use Claude 4 Sonnet to invent a dog: ```bash llm -m claude-sonnet-5 --schema 'name,age int,bio: one sentence' 'invent a surprising dog' ``` Example output: ```json { "name": "Whiskers the Mathematical Mastiff", "age": 7, "bio": "Whiskers is a mastiff who can solve complex calculus problems by barking in binary code and has won three international mathematics competitions against human competitors." } ``` ## Web search Newer models support Anthropic's [web search tool](https://platform.claude.com/docs/en/agents-and-tools/tool-use/web-search-tool) for real-time information, using the `-T WebSearch` server-side tool: ```bash llm -m claude-sonnet-5 -T WebSearch 'What is the current weather in San Francisco?' ``` The tool accepts optional configuration: ```bash llm -m claude-sonnet-5 \ -T 'WebSearch(max_uses=2, user_location={"city": "London", "country": "GB"})' \ 'Recent headlines' ``` Available arguments: - `max_uses`: maximum number of searches per request - `allowed_domains` / `blocked_domains`: lists of domains to allow or block (cannot be combined) - `user_location`: dictionary with optional `city`, `region`, `country` and `timezone` keys to localize results Note that `user_location` affects the *results* of searches from the web search tool, but location information is not made directly available to the model. On Claude 4.6 and later models this uses the `web_search_20260318` tool version with dynamic filtering; older models use the basic `web_search_20250305` version. ## Web fetch Models that support web search can also use Anthropic's [web fetch tool](https://platform.claude.com/docs/en/agents-and-tools/tool-use/web-fetch-tool) to retrieve the full content of a URL, using the `-T WebFetch` server-side tool: ```bash llm -m claude-sonnet-5 -T WebFetch 'Fetch https://www.example.com/ and quote its first heading' ``` For security reasons Claude can only fetch URLs that already appear in the conversation - provided by you or returned by a previous web search or fetch. The tool accepts optional configuration: ```bash llm -m claude-sonnet-5 \ -T 'WebFetch(max_uses=2, max_content_tokens=20000)' \ 'Summarize https://www.example.com/' ``` Available arguments: - `max_uses`: maximum number of fetches per request - `allowed_domains` / `blocked_domains`: lists of domains to allow or block (cannot be combined) - `citations`: set to `True` to enable citations for fetched content - `max_content_tokens`: approximate cap on fetched content included in the context - `use_cache`: set to `False` to bypass Anthropic's fetch cache (Claude 4.6 and later models only) On Claude 4.6 and later models this uses the `web_fetch_20260318` tool version with [dynamic filtering](https://platform.claude.com/docs/en/agents-and-tools/tool-use/web-fetch-tool#dynamic-filtering); older models use the basic `web_fetch_20250910` version. From Python, pass an instance of the `WebFetch` class in `tools=`: ```python import llm from llm_anthropic import WebFetch model = llm.get_model("claude-sonnet-5") response = model.prompt( "Fetch https://www.example.com/ and quote its first heading", tools=[WebFetch(max_uses=1)], ) print(response.text()) ``` ## MCP connector Models that support web search can also call tools on remote [MCP servers](https://platform.claude.com/docs/en/agents-and-tools/mcp-connector) using the `AnthropicMCP` server-side tool. Anthropic connects to the server from their own infrastructure - it must be reachable over HTTPS: ```bash llm -m claude-sonnet-5 \ -T 'AnthropicMCP(url="https://mcp.deepwiki.com/mcp", name="deepwiki")' \ 'Use the deepwiki tools to say what simonw/llm does, one sentence' ``` Available arguments: - `url`: the HTTPS URL of the remote MCP server (required) - `name`: an identifier for the server - defaults to the URL's hostname - `authorization_token`: OAuth bearer token, for servers that require authentication - `allowed_tools`: optional list of tool names - if provided, only those tools are enabled ```bash llm -m claude-sonnet-5 \ -T 'AnthropicMCP(url="https://mcp.deepwiki.com/mcp", name="deepwiki", allowed_tools=["ask_question"])' \ 'What does simonw/llm do?' ``` You can pass multiple `MCP` tools to connect to more than one server in the same request. Only MCP tool calls are supported - not MCP resources or prompts. From Python, pass an instance of the `AnthropicMCP` class in `tools=`: ```python import llm from llm_anthropic import AnthropicMCP model = llm.get_model("claude-sonnet-5") response = model.prompt( "Use the deepwiki tools to say what simonw/llm does, one sentence", tools=[AnthropicMCP(url="https://mcp.deepwiki.com/mcp", name="deepwiki")], ) print(response.text()) ``` This feature uses Anthropic's `mcp-client-2025-11-20` beta. ## Code execution Claude 4.5 and later models support Anthropic's [code execution tool](https://platform.claude.com/docs/en/agents-and-tools/tool-use/code-execution-tool), which runs Python and bash in a sandboxed server-side container. Use the `-T CodeExecution` server-side tool: ```bash llm -m claude-sonnet-4.6 -T CodeExecution \ 'Compute the sha256 hex digest of the string "pelican"' ``` Each response that runs code reports a container ID in its `response_json`, visible with `llm logs --json`. Pass that ID back to reuse the container's files and state in a later prompt (containers expire after a period of inactivity): ```bash llm -m claude-sonnet-4.6 -T 'CodeExecution(container="container_011CPd...")' \ 'Read /tmp/results.csv and summarize it' ``` ## Fast mode Some models support [fast mode](https://platform.claude.com/docs/en/build-with-claude/fast-mode) for lower latency responses. Enable it with the `-o fast 1` option: ```bash llm -m claude-opus-5 -o fast 1 'Fun facts about walruses' ``` ## Usage from Python Python code can access the models like this: ```python import llm model = llm.get_model("claude-haiku-4.5") print(model.prompt("Fun facts about chipmunks")) ``` Consult [LLM's Python API documentation](https://llm.datasette.io/en/stable/python-api.html) for more details. You can also import the model classes directly, which is useful if you want to point the `base_url` at a different Anthropic-compatible endpoint: ```python from llm_anthropic import ClaudeMessages model = ClaudeMessages( "MiniMax-M2", base_url="https://api.minimax.io/anthropic" ) print(model.prompt("Fun facts about pangolins", key="eyJh...")) ``` ## Extended thinking Anthropic models can spend [thinking tokens](https://platform.claude.com/docs/en/build-with-claude/extended-thinking) reasoning through a prompt before producing their response. LLM streams that reasoning to standard error as it arrives - pass `-R/--hide-reasoning` to hide it. The reasoning is also logged, available as the `reasoning` field in `llm logs --json`. **Claude 5 models think by default.** Tune how hard they think with the `thinking_effort` option - one of `low`, `medium`, `high`, `xhigh` or `max`: ```bash llm -m claude-opus-5 -o thinking_effort max 'Design a fair algorithm for splitting rent between roommates with different sized rooms' ``` Sonnet 5 and Opus 5 can have thinking turned off entirely with `-o thinking 0`. Fable 5 always thinks - disabling it raises an error. **Claude 4.6 and older models do not think unless asked.** Enable thinking with `-o thinking 1`: ```bash llm -m claude-sonnet-4.6 -o thinking 1 'Write a convincing speech to congress about the need to protect the California Brown Pelican' ``` Claude 4.6 models (and Opus 4.5) also support `thinking_effort`, which implies `thinking 1`. Older models than that use a fixed 1,024 token thinking budget. When `-R/--hide-reasoning` is set this plugin also passes `display: omitted` to the Anthropic API, which leaves the thinking trace out of the response entirely - it will not appear in your logs, though thinking tokens are still billed. The `thinking_budget`, `thinking_display` and `thinking_adaptive` options were removed in llm-anthropic 0.26 - install `llm-anthropic==0.25` if you need them for older models. ## Model options The following options can be passed using `-o name value` on the CLI or as `keyword=value` arguments to the Python `model.prompt()` method: - **max_tokens**: `int` The maximum number of tokens to generate before stopping - **temperature**: `float` Amount of randomness injected into the response. Defaults to 1.0. Ranges from 0.0 to 1.0. Use temperature closer to 0.0 for analytical / multiple choice, and closer to 1.0 for creative and generative tasks. Note that even with temperature of 0.0, the results will not be fully deterministic. - **top_p**: `float` Use nucleus sampling. In nucleus sampling, we compute the cumulative distribution over all the options for each subsequent token in decreasing probability order and cut it off once it reaches a particular probability specified by top_p. You should either alter temperature or top_p, but not both. Recommended for advanced use cases only. You usually only need to use temperature. - **top_k**: `int` Only sample from the top K options for each subsequent token. Used to remove 'long tail' low probability responses. Recommended for advanced use cases only. You usually only need to use temperature. - **user_id**: `str` An external identifier for the user who is associated with the request - **prefill**: `str` A prefill to use for the response - **hide_prefill**: `boolean` Do not repeat the prefill value at the start of the response - **stop_sequences**: `array, str` Custom text sequences that will cause the model to stop generating - pass either a list of strings or a single string - **cache**: `boolean` Use Anthropic prompt cache for any attachments or fragments - **fast**: `boolean` Use fast mode for lower latency responses: https://platform.claude.com/docs/en/build-with-claude/fast-mode - **thinking**: `boolean` Enable thinking mode. Claude 5 models think by default - set to false to disable thinking on models that allow it The `prefill` option can be used to set the first part of the response. To increase the chance of returning JSON, set that to `{`: ```bash llm -m claude-sonnet-5 'Fun data about pelicans' \ -o prefill '{' ``` If you do not want the prefill token to be echoed in the response, set `hide_prefill` to `true`: ```bash llm -m claude-haiku-4.5 'Short python function describing a pelican' \ -o prefill '```python' \ -o hide_prefill true \ -o stop_sequences '```' ``` This example sets `` ``` `` as the stop sequence, so the response will be a Python function without the wrapping Markdown code block. To pass a single stop sequence, send a string: ```bash llm -m claude-sonnet-5 'Fun facts about pelicans' \ -o stop-sequences "beak" ``` For multiple stop sequences, pass a JSON array: ```bash llm -m claude-sonnet-5 'Fun facts about pelicans' \ -o stop-sequences '["beak", "feathers"]' ``` When using the Python API, pass a string or an array of strings: ```python response = llm.query( model="claude-sonnet-5", query="Fun facts about pelicans", stop_sequences=["beak", "feathers"], ) ``` ## Development To set up this plugin locally, first checkout the code. Then create a new virtual environment: ```bash cd llm-anthropic python3 -m venv venv source venv/bin/activate ``` Now install the dependencies and test dependencies: ```bash pip install -e . --group dev ``` To run the tests: ```bash pytest ``` Alternatively, if you have [uv](https://github.com/astral-sh/uv) you can run tests without first creating a virtual environment like this: ```bash uv run pytest uv run pytest -k test_tools ``` You can also run the `llm` command in a `uv` managed environment like this: ```bash uv run llm 'your prompt here' ``` To enable debug logs while running ([like this](https://github.com/simonw/llm-anthropic/issues/54#issuecomment-3536842831)), set this environment variable: ```bash export ANTHROPIC_LOG=debug ``` This project uses [pytest-recording](https://github.com/kiwicom/pytest-recording) to record Anthropic API responses for the tests, and [inline-snapshot](https://15r10nk.github.io/inline-snapshot/) for test assertions. If you add a new test that calls the API you can capture the API response like this: ```bash PYTEST_ANTHROPIC_API_KEY="$(llm keys get anthropic)" uv run pytest --record-mode once ``` You will need to have stored a valid Anthropic API key using this command first: ```bash llm keys set anthropic # Paste key here ``` To re-record all cassettes and update all inline snapshot assertions in one command: ```bash rm tests/cassettes/test_anthropic/*.yaml PYTEST_ANTHROPIC_API_KEY="$(llm keys get anthropic)" uv run pytest --record-mode all --inline-snapshot=fix ``` To re-record a single test: ```bash rm tests/cassettes/test_anthropic/test_thinking_prompt.yaml PYTEST_ANTHROPIC_API_KEY="$(llm keys get anthropic)" uv run pytest -k test_thinking_prompt --record-mode once --inline-snapshot=fix ``` ././@PaxHeader0000000000000000000000000000002600000000000010213 xustar0022 mtime=1785880929.0 llm_anthropic-0.26/README.md0000644000175100017510000003726215234460541015237 0ustar00runnerrunner# llm-anthropic [![PyPI](https://img.shields.io/pypi/v/llm-anthropic.svg)](https://pypi.org/project/llm-anthropic/) [![Changelog](https://img.shields.io/github/v/release/simonw/llm-anthropic?include_prereleases&label=changelog)](https://github.com/simonw/llm-anthropic/releases) [![Tests](https://github.com/simonw/llm-anthropic/actions/workflows/test.yml/badge.svg)](https://github.com/simonw/llm-anthropic/actions/workflows/test.yml) [![License](https://img.shields.io/badge/license-Apache%202.0-blue.svg)](https://github.com/simonw/llm-anthropic/blob/main/LICENSE) LLM access to models by Anthropic, including the Claude series ## Installation Install this plugin in the same environment as [LLM](https://llm.datasette.io/). ```bash llm install llm-anthropic ```
Instructions for users who need to upgrade from llm-claude-3
If you previously used `llm-claude-3` you can upgrade like this: ```bash llm install -U llm-claude-3 llm keys set anthropic --value "$(llm keys get claude)" ``` The first line will remove the previous `llm-claude-3` version and install this one, because the latest `llm-claude-3` depends on `llm-anthropic`. The second line sets the `anthropic` key to whatever value you previously used for the `claude` key.
## Usage First, set [an API key](https://console.anthropic.com/settings/keys) for Anthropic: ```bash llm keys set anthropic # Paste key here ``` You can also set the key in the environment variable `ANTHROPIC_API_KEY` Run `llm models` to list the models, and `llm models --options` to include a list of their options. Run prompts like this: ```bash llm -m claude-opus-5 'Fun facts about walruses' llm -m claude-sonnet-5 'Fun facts about pelicans' llm -m claude-haiku-4.5 'Fun facts about cormorants' ``` Image attachments are supported too: ```bash llm -m claude-sonnet-5 'describe this image' -a https://static.simonwillison.net/static/2024/pelicans.jpg llm -m claude-haiku-4.5 'extract text' -a page.png ``` The Claude 3.5 and 4 models can handle PDF files: ```bash llm -m claude-sonnet-5 'extract text' -a page.pdf ``` Anthropic's models support [schemas](https://llm.datasette.io/en/stable/schemas.html). Here's how to use Claude 4 Sonnet to invent a dog: ```bash llm -m claude-sonnet-5 --schema 'name,age int,bio: one sentence' 'invent a surprising dog' ``` Example output: ```json { "name": "Whiskers the Mathematical Mastiff", "age": 7, "bio": "Whiskers is a mastiff who can solve complex calculus problems by barking in binary code and has won three international mathematics competitions against human competitors." } ``` ## Web search Newer models support Anthropic's [web search tool](https://platform.claude.com/docs/en/agents-and-tools/tool-use/web-search-tool) for real-time information, using the `-T WebSearch` server-side tool: ```bash llm -m claude-sonnet-5 -T WebSearch 'What is the current weather in San Francisco?' ``` The tool accepts optional configuration: ```bash llm -m claude-sonnet-5 \ -T 'WebSearch(max_uses=2, user_location={"city": "London", "country": "GB"})' \ 'Recent headlines' ``` Available arguments: - `max_uses`: maximum number of searches per request - `allowed_domains` / `blocked_domains`: lists of domains to allow or block (cannot be combined) - `user_location`: dictionary with optional `city`, `region`, `country` and `timezone` keys to localize results Note that `user_location` affects the *results* of searches from the web search tool, but location information is not made directly available to the model. On Claude 4.6 and later models this uses the `web_search_20260318` tool version with dynamic filtering; older models use the basic `web_search_20250305` version. ## Web fetch Models that support web search can also use Anthropic's [web fetch tool](https://platform.claude.com/docs/en/agents-and-tools/tool-use/web-fetch-tool) to retrieve the full content of a URL, using the `-T WebFetch` server-side tool: ```bash llm -m claude-sonnet-5 -T WebFetch 'Fetch https://www.example.com/ and quote its first heading' ``` For security reasons Claude can only fetch URLs that already appear in the conversation - provided by you or returned by a previous web search or fetch. The tool accepts optional configuration: ```bash llm -m claude-sonnet-5 \ -T 'WebFetch(max_uses=2, max_content_tokens=20000)' \ 'Summarize https://www.example.com/' ``` Available arguments: - `max_uses`: maximum number of fetches per request - `allowed_domains` / `blocked_domains`: lists of domains to allow or block (cannot be combined) - `citations`: set to `True` to enable citations for fetched content - `max_content_tokens`: approximate cap on fetched content included in the context - `use_cache`: set to `False` to bypass Anthropic's fetch cache (Claude 4.6 and later models only) On Claude 4.6 and later models this uses the `web_fetch_20260318` tool version with [dynamic filtering](https://platform.claude.com/docs/en/agents-and-tools/tool-use/web-fetch-tool#dynamic-filtering); older models use the basic `web_fetch_20250910` version. From Python, pass an instance of the `WebFetch` class in `tools=`: ```python import llm from llm_anthropic import WebFetch model = llm.get_model("claude-sonnet-5") response = model.prompt( "Fetch https://www.example.com/ and quote its first heading", tools=[WebFetch(max_uses=1)], ) print(response.text()) ``` ## MCP connector Models that support web search can also call tools on remote [MCP servers](https://platform.claude.com/docs/en/agents-and-tools/mcp-connector) using the `AnthropicMCP` server-side tool. Anthropic connects to the server from their own infrastructure - it must be reachable over HTTPS: ```bash llm -m claude-sonnet-5 \ -T 'AnthropicMCP(url="https://mcp.deepwiki.com/mcp", name="deepwiki")' \ 'Use the deepwiki tools to say what simonw/llm does, one sentence' ``` Available arguments: - `url`: the HTTPS URL of the remote MCP server (required) - `name`: an identifier for the server - defaults to the URL's hostname - `authorization_token`: OAuth bearer token, for servers that require authentication - `allowed_tools`: optional list of tool names - if provided, only those tools are enabled ```bash llm -m claude-sonnet-5 \ -T 'AnthropicMCP(url="https://mcp.deepwiki.com/mcp", name="deepwiki", allowed_tools=["ask_question"])' \ 'What does simonw/llm do?' ``` You can pass multiple `MCP` tools to connect to more than one server in the same request. Only MCP tool calls are supported - not MCP resources or prompts. From Python, pass an instance of the `AnthropicMCP` class in `tools=`: ```python import llm from llm_anthropic import AnthropicMCP model = llm.get_model("claude-sonnet-5") response = model.prompt( "Use the deepwiki tools to say what simonw/llm does, one sentence", tools=[AnthropicMCP(url="https://mcp.deepwiki.com/mcp", name="deepwiki")], ) print(response.text()) ``` This feature uses Anthropic's `mcp-client-2025-11-20` beta. ## Code execution Claude 4.5 and later models support Anthropic's [code execution tool](https://platform.claude.com/docs/en/agents-and-tools/tool-use/code-execution-tool), which runs Python and bash in a sandboxed server-side container. Use the `-T CodeExecution` server-side tool: ```bash llm -m claude-sonnet-4.6 -T CodeExecution \ 'Compute the sha256 hex digest of the string "pelican"' ``` Each response that runs code reports a container ID in its `response_json`, visible with `llm logs --json`. Pass that ID back to reuse the container's files and state in a later prompt (containers expire after a period of inactivity): ```bash llm -m claude-sonnet-4.6 -T 'CodeExecution(container="container_011CPd...")' \ 'Read /tmp/results.csv and summarize it' ``` ## Fast mode Some models support [fast mode](https://platform.claude.com/docs/en/build-with-claude/fast-mode) for lower latency responses. Enable it with the `-o fast 1` option: ```bash llm -m claude-opus-5 -o fast 1 'Fun facts about walruses' ``` ## Usage from Python Python code can access the models like this: ```python import llm model = llm.get_model("claude-haiku-4.5") print(model.prompt("Fun facts about chipmunks")) ``` Consult [LLM's Python API documentation](https://llm.datasette.io/en/stable/python-api.html) for more details. You can also import the model classes directly, which is useful if you want to point the `base_url` at a different Anthropic-compatible endpoint: ```python from llm_anthropic import ClaudeMessages model = ClaudeMessages( "MiniMax-M2", base_url="https://api.minimax.io/anthropic" ) print(model.prompt("Fun facts about pangolins", key="eyJh...")) ``` ## Extended thinking Anthropic models can spend [thinking tokens](https://platform.claude.com/docs/en/build-with-claude/extended-thinking) reasoning through a prompt before producing their response. LLM streams that reasoning to standard error as it arrives - pass `-R/--hide-reasoning` to hide it. The reasoning is also logged, available as the `reasoning` field in `llm logs --json`. **Claude 5 models think by default.** Tune how hard they think with the `thinking_effort` option - one of `low`, `medium`, `high`, `xhigh` or `max`: ```bash llm -m claude-opus-5 -o thinking_effort max 'Design a fair algorithm for splitting rent between roommates with different sized rooms' ``` Sonnet 5 and Opus 5 can have thinking turned off entirely with `-o thinking 0`. Fable 5 always thinks - disabling it raises an error. **Claude 4.6 and older models do not think unless asked.** Enable thinking with `-o thinking 1`: ```bash llm -m claude-sonnet-4.6 -o thinking 1 'Write a convincing speech to congress about the need to protect the California Brown Pelican' ``` Claude 4.6 models (and Opus 4.5) also support `thinking_effort`, which implies `thinking 1`. Older models than that use a fixed 1,024 token thinking budget. When `-R/--hide-reasoning` is set this plugin also passes `display: omitted` to the Anthropic API, which leaves the thinking trace out of the response entirely - it will not appear in your logs, though thinking tokens are still billed. The `thinking_budget`, `thinking_display` and `thinking_adaptive` options were removed in llm-anthropic 0.26 - install `llm-anthropic==0.25` if you need them for older models. ## Model options The following options can be passed using `-o name value` on the CLI or as `keyword=value` arguments to the Python `model.prompt()` method: - **max_tokens**: `int` The maximum number of tokens to generate before stopping - **temperature**: `float` Amount of randomness injected into the response. Defaults to 1.0. Ranges from 0.0 to 1.0. Use temperature closer to 0.0 for analytical / multiple choice, and closer to 1.0 for creative and generative tasks. Note that even with temperature of 0.0, the results will not be fully deterministic. - **top_p**: `float` Use nucleus sampling. In nucleus sampling, we compute the cumulative distribution over all the options for each subsequent token in decreasing probability order and cut it off once it reaches a particular probability specified by top_p. You should either alter temperature or top_p, but not both. Recommended for advanced use cases only. You usually only need to use temperature. - **top_k**: `int` Only sample from the top K options for each subsequent token. Used to remove 'long tail' low probability responses. Recommended for advanced use cases only. You usually only need to use temperature. - **user_id**: `str` An external identifier for the user who is associated with the request - **prefill**: `str` A prefill to use for the response - **hide_prefill**: `boolean` Do not repeat the prefill value at the start of the response - **stop_sequences**: `array, str` Custom text sequences that will cause the model to stop generating - pass either a list of strings or a single string - **cache**: `boolean` Use Anthropic prompt cache for any attachments or fragments - **fast**: `boolean` Use fast mode for lower latency responses: https://platform.claude.com/docs/en/build-with-claude/fast-mode - **thinking**: `boolean` Enable thinking mode. Claude 5 models think by default - set to false to disable thinking on models that allow it The `prefill` option can be used to set the first part of the response. To increase the chance of returning JSON, set that to `{`: ```bash llm -m claude-sonnet-5 'Fun data about pelicans' \ -o prefill '{' ``` If you do not want the prefill token to be echoed in the response, set `hide_prefill` to `true`: ```bash llm -m claude-haiku-4.5 'Short python function describing a pelican' \ -o prefill '```python' \ -o hide_prefill true \ -o stop_sequences '```' ``` This example sets `` ``` `` as the stop sequence, so the response will be a Python function without the wrapping Markdown code block. To pass a single stop sequence, send a string: ```bash llm -m claude-sonnet-5 'Fun facts about pelicans' \ -o stop-sequences "beak" ``` For multiple stop sequences, pass a JSON array: ```bash llm -m claude-sonnet-5 'Fun facts about pelicans' \ -o stop-sequences '["beak", "feathers"]' ``` When using the Python API, pass a string or an array of strings: ```python response = llm.query( model="claude-sonnet-5", query="Fun facts about pelicans", stop_sequences=["beak", "feathers"], ) ``` ## Development To set up this plugin locally, first checkout the code. Then create a new virtual environment: ```bash cd llm-anthropic python3 -m venv venv source venv/bin/activate ``` Now install the dependencies and test dependencies: ```bash pip install -e . --group dev ``` To run the tests: ```bash pytest ``` Alternatively, if you have [uv](https://github.com/astral-sh/uv) you can run tests without first creating a virtual environment like this: ```bash uv run pytest uv run pytest -k test_tools ``` You can also run the `llm` command in a `uv` managed environment like this: ```bash uv run llm 'your prompt here' ``` To enable debug logs while running ([like this](https://github.com/simonw/llm-anthropic/issues/54#issuecomment-3536842831)), set this environment variable: ```bash export ANTHROPIC_LOG=debug ``` This project uses [pytest-recording](https://github.com/kiwicom/pytest-recording) to record Anthropic API responses for the tests, and [inline-snapshot](https://15r10nk.github.io/inline-snapshot/) for test assertions. If you add a new test that calls the API you can capture the API response like this: ```bash PYTEST_ANTHROPIC_API_KEY="$(llm keys get anthropic)" uv run pytest --record-mode once ``` You will need to have stored a valid Anthropic API key using this command first: ```bash llm keys set anthropic # Paste key here ``` To re-record all cassettes and update all inline snapshot assertions in one command: ```bash rm tests/cassettes/test_anthropic/*.yaml PYTEST_ANTHROPIC_API_KEY="$(llm keys get anthropic)" uv run pytest --record-mode all --inline-snapshot=fix ``` To re-record a single test: ```bash rm tests/cassettes/test_anthropic/test_thinking_prompt.yaml PYTEST_ANTHROPIC_API_KEY="$(llm keys get anthropic)" uv run pytest -k test_thinking_prompt --record-mode once --inline-snapshot=fix ``` ././@PaxHeader0000000000000000000000000000003400000000000010212 xustar0028 mtime=1785880938.9591715 llm_anthropic-0.26/llm_anthropic.egg-info/0000755000175100017510000000000015234460553020276 5ustar00runnerrunner././@PaxHeader0000000000000000000000000000002600000000000010213 xustar0022 mtime=1785880938.0 llm_anthropic-0.26/llm_anthropic.egg-info/PKG-INFO0000644000175100017510000004043415234460552021377 0ustar00runnerrunnerMetadata-Version: 2.4 Name: llm-anthropic Version: 0.26 Summary: LLM access to models by Anthropic, including the Claude series Author: Simon Willison License-Expression: Apache-2.0 Project-URL: Homepage, https://github.com/simonw/llm-anthropic Project-URL: Changelog, https://github.com/simonw/llm-anthropic/releases Project-URL: Issues, https://github.com/simonw/llm-anthropic/issues Project-URL: CI, https://github.com/simonw/llm-anthropic/actions Requires-Python: >=3.10 Description-Content-Type: text/markdown License-File: LICENSE Requires-Dist: llm>=0.32 Requires-Dist: anthropic>=0.96.0 Dynamic: license-file # llm-anthropic [![PyPI](https://img.shields.io/pypi/v/llm-anthropic.svg)](https://pypi.org/project/llm-anthropic/) [![Changelog](https://img.shields.io/github/v/release/simonw/llm-anthropic?include_prereleases&label=changelog)](https://github.com/simonw/llm-anthropic/releases) [![Tests](https://github.com/simonw/llm-anthropic/actions/workflows/test.yml/badge.svg)](https://github.com/simonw/llm-anthropic/actions/workflows/test.yml) [![License](https://img.shields.io/badge/license-Apache%202.0-blue.svg)](https://github.com/simonw/llm-anthropic/blob/main/LICENSE) LLM access to models by Anthropic, including the Claude series ## Installation Install this plugin in the same environment as [LLM](https://llm.datasette.io/). ```bash llm install llm-anthropic ```
Instructions for users who need to upgrade from llm-claude-3
If you previously used `llm-claude-3` you can upgrade like this: ```bash llm install -U llm-claude-3 llm keys set anthropic --value "$(llm keys get claude)" ``` The first line will remove the previous `llm-claude-3` version and install this one, because the latest `llm-claude-3` depends on `llm-anthropic`. The second line sets the `anthropic` key to whatever value you previously used for the `claude` key.
## Usage First, set [an API key](https://console.anthropic.com/settings/keys) for Anthropic: ```bash llm keys set anthropic # Paste key here ``` You can also set the key in the environment variable `ANTHROPIC_API_KEY` Run `llm models` to list the models, and `llm models --options` to include a list of their options. Run prompts like this: ```bash llm -m claude-opus-5 'Fun facts about walruses' llm -m claude-sonnet-5 'Fun facts about pelicans' llm -m claude-haiku-4.5 'Fun facts about cormorants' ``` Image attachments are supported too: ```bash llm -m claude-sonnet-5 'describe this image' -a https://static.simonwillison.net/static/2024/pelicans.jpg llm -m claude-haiku-4.5 'extract text' -a page.png ``` The Claude 3.5 and 4 models can handle PDF files: ```bash llm -m claude-sonnet-5 'extract text' -a page.pdf ``` Anthropic's models support [schemas](https://llm.datasette.io/en/stable/schemas.html). Here's how to use Claude 4 Sonnet to invent a dog: ```bash llm -m claude-sonnet-5 --schema 'name,age int,bio: one sentence' 'invent a surprising dog' ``` Example output: ```json { "name": "Whiskers the Mathematical Mastiff", "age": 7, "bio": "Whiskers is a mastiff who can solve complex calculus problems by barking in binary code and has won three international mathematics competitions against human competitors." } ``` ## Web search Newer models support Anthropic's [web search tool](https://platform.claude.com/docs/en/agents-and-tools/tool-use/web-search-tool) for real-time information, using the `-T WebSearch` server-side tool: ```bash llm -m claude-sonnet-5 -T WebSearch 'What is the current weather in San Francisco?' ``` The tool accepts optional configuration: ```bash llm -m claude-sonnet-5 \ -T 'WebSearch(max_uses=2, user_location={"city": "London", "country": "GB"})' \ 'Recent headlines' ``` Available arguments: - `max_uses`: maximum number of searches per request - `allowed_domains` / `blocked_domains`: lists of domains to allow or block (cannot be combined) - `user_location`: dictionary with optional `city`, `region`, `country` and `timezone` keys to localize results Note that `user_location` affects the *results* of searches from the web search tool, but location information is not made directly available to the model. On Claude 4.6 and later models this uses the `web_search_20260318` tool version with dynamic filtering; older models use the basic `web_search_20250305` version. ## Web fetch Models that support web search can also use Anthropic's [web fetch tool](https://platform.claude.com/docs/en/agents-and-tools/tool-use/web-fetch-tool) to retrieve the full content of a URL, using the `-T WebFetch` server-side tool: ```bash llm -m claude-sonnet-5 -T WebFetch 'Fetch https://www.example.com/ and quote its first heading' ``` For security reasons Claude can only fetch URLs that already appear in the conversation - provided by you or returned by a previous web search or fetch. The tool accepts optional configuration: ```bash llm -m claude-sonnet-5 \ -T 'WebFetch(max_uses=2, max_content_tokens=20000)' \ 'Summarize https://www.example.com/' ``` Available arguments: - `max_uses`: maximum number of fetches per request - `allowed_domains` / `blocked_domains`: lists of domains to allow or block (cannot be combined) - `citations`: set to `True` to enable citations for fetched content - `max_content_tokens`: approximate cap on fetched content included in the context - `use_cache`: set to `False` to bypass Anthropic's fetch cache (Claude 4.6 and later models only) On Claude 4.6 and later models this uses the `web_fetch_20260318` tool version with [dynamic filtering](https://platform.claude.com/docs/en/agents-and-tools/tool-use/web-fetch-tool#dynamic-filtering); older models use the basic `web_fetch_20250910` version. From Python, pass an instance of the `WebFetch` class in `tools=`: ```python import llm from llm_anthropic import WebFetch model = llm.get_model("claude-sonnet-5") response = model.prompt( "Fetch https://www.example.com/ and quote its first heading", tools=[WebFetch(max_uses=1)], ) print(response.text()) ``` ## MCP connector Models that support web search can also call tools on remote [MCP servers](https://platform.claude.com/docs/en/agents-and-tools/mcp-connector) using the `AnthropicMCP` server-side tool. Anthropic connects to the server from their own infrastructure - it must be reachable over HTTPS: ```bash llm -m claude-sonnet-5 \ -T 'AnthropicMCP(url="https://mcp.deepwiki.com/mcp", name="deepwiki")' \ 'Use the deepwiki tools to say what simonw/llm does, one sentence' ``` Available arguments: - `url`: the HTTPS URL of the remote MCP server (required) - `name`: an identifier for the server - defaults to the URL's hostname - `authorization_token`: OAuth bearer token, for servers that require authentication - `allowed_tools`: optional list of tool names - if provided, only those tools are enabled ```bash llm -m claude-sonnet-5 \ -T 'AnthropicMCP(url="https://mcp.deepwiki.com/mcp", name="deepwiki", allowed_tools=["ask_question"])' \ 'What does simonw/llm do?' ``` You can pass multiple `MCP` tools to connect to more than one server in the same request. Only MCP tool calls are supported - not MCP resources or prompts. From Python, pass an instance of the `AnthropicMCP` class in `tools=`: ```python import llm from llm_anthropic import AnthropicMCP model = llm.get_model("claude-sonnet-5") response = model.prompt( "Use the deepwiki tools to say what simonw/llm does, one sentence", tools=[AnthropicMCP(url="https://mcp.deepwiki.com/mcp", name="deepwiki")], ) print(response.text()) ``` This feature uses Anthropic's `mcp-client-2025-11-20` beta. ## Code execution Claude 4.5 and later models support Anthropic's [code execution tool](https://platform.claude.com/docs/en/agents-and-tools/tool-use/code-execution-tool), which runs Python and bash in a sandboxed server-side container. Use the `-T CodeExecution` server-side tool: ```bash llm -m claude-sonnet-4.6 -T CodeExecution \ 'Compute the sha256 hex digest of the string "pelican"' ``` Each response that runs code reports a container ID in its `response_json`, visible with `llm logs --json`. Pass that ID back to reuse the container's files and state in a later prompt (containers expire after a period of inactivity): ```bash llm -m claude-sonnet-4.6 -T 'CodeExecution(container="container_011CPd...")' \ 'Read /tmp/results.csv and summarize it' ``` ## Fast mode Some models support [fast mode](https://platform.claude.com/docs/en/build-with-claude/fast-mode) for lower latency responses. Enable it with the `-o fast 1` option: ```bash llm -m claude-opus-5 -o fast 1 'Fun facts about walruses' ``` ## Usage from Python Python code can access the models like this: ```python import llm model = llm.get_model("claude-haiku-4.5") print(model.prompt("Fun facts about chipmunks")) ``` Consult [LLM's Python API documentation](https://llm.datasette.io/en/stable/python-api.html) for more details. You can also import the model classes directly, which is useful if you want to point the `base_url` at a different Anthropic-compatible endpoint: ```python from llm_anthropic import ClaudeMessages model = ClaudeMessages( "MiniMax-M2", base_url="https://api.minimax.io/anthropic" ) print(model.prompt("Fun facts about pangolins", key="eyJh...")) ``` ## Extended thinking Anthropic models can spend [thinking tokens](https://platform.claude.com/docs/en/build-with-claude/extended-thinking) reasoning through a prompt before producing their response. LLM streams that reasoning to standard error as it arrives - pass `-R/--hide-reasoning` to hide it. The reasoning is also logged, available as the `reasoning` field in `llm logs --json`. **Claude 5 models think by default.** Tune how hard they think with the `thinking_effort` option - one of `low`, `medium`, `high`, `xhigh` or `max`: ```bash llm -m claude-opus-5 -o thinking_effort max 'Design a fair algorithm for splitting rent between roommates with different sized rooms' ``` Sonnet 5 and Opus 5 can have thinking turned off entirely with `-o thinking 0`. Fable 5 always thinks - disabling it raises an error. **Claude 4.6 and older models do not think unless asked.** Enable thinking with `-o thinking 1`: ```bash llm -m claude-sonnet-4.6 -o thinking 1 'Write a convincing speech to congress about the need to protect the California Brown Pelican' ``` Claude 4.6 models (and Opus 4.5) also support `thinking_effort`, which implies `thinking 1`. Older models than that use a fixed 1,024 token thinking budget. When `-R/--hide-reasoning` is set this plugin also passes `display: omitted` to the Anthropic API, which leaves the thinking trace out of the response entirely - it will not appear in your logs, though thinking tokens are still billed. The `thinking_budget`, `thinking_display` and `thinking_adaptive` options were removed in llm-anthropic 0.26 - install `llm-anthropic==0.25` if you need them for older models. ## Model options The following options can be passed using `-o name value` on the CLI or as `keyword=value` arguments to the Python `model.prompt()` method: - **max_tokens**: `int` The maximum number of tokens to generate before stopping - **temperature**: `float` Amount of randomness injected into the response. Defaults to 1.0. Ranges from 0.0 to 1.0. Use temperature closer to 0.0 for analytical / multiple choice, and closer to 1.0 for creative and generative tasks. Note that even with temperature of 0.0, the results will not be fully deterministic. - **top_p**: `float` Use nucleus sampling. In nucleus sampling, we compute the cumulative distribution over all the options for each subsequent token in decreasing probability order and cut it off once it reaches a particular probability specified by top_p. You should either alter temperature or top_p, but not both. Recommended for advanced use cases only. You usually only need to use temperature. - **top_k**: `int` Only sample from the top K options for each subsequent token. Used to remove 'long tail' low probability responses. Recommended for advanced use cases only. You usually only need to use temperature. - **user_id**: `str` An external identifier for the user who is associated with the request - **prefill**: `str` A prefill to use for the response - **hide_prefill**: `boolean` Do not repeat the prefill value at the start of the response - **stop_sequences**: `array, str` Custom text sequences that will cause the model to stop generating - pass either a list of strings or a single string - **cache**: `boolean` Use Anthropic prompt cache for any attachments or fragments - **fast**: `boolean` Use fast mode for lower latency responses: https://platform.claude.com/docs/en/build-with-claude/fast-mode - **thinking**: `boolean` Enable thinking mode. Claude 5 models think by default - set to false to disable thinking on models that allow it The `prefill` option can be used to set the first part of the response. To increase the chance of returning JSON, set that to `{`: ```bash llm -m claude-sonnet-5 'Fun data about pelicans' \ -o prefill '{' ``` If you do not want the prefill token to be echoed in the response, set `hide_prefill` to `true`: ```bash llm -m claude-haiku-4.5 'Short python function describing a pelican' \ -o prefill '```python' \ -o hide_prefill true \ -o stop_sequences '```' ``` This example sets `` ``` `` as the stop sequence, so the response will be a Python function without the wrapping Markdown code block. To pass a single stop sequence, send a string: ```bash llm -m claude-sonnet-5 'Fun facts about pelicans' \ -o stop-sequences "beak" ``` For multiple stop sequences, pass a JSON array: ```bash llm -m claude-sonnet-5 'Fun facts about pelicans' \ -o stop-sequences '["beak", "feathers"]' ``` When using the Python API, pass a string or an array of strings: ```python response = llm.query( model="claude-sonnet-5", query="Fun facts about pelicans", stop_sequences=["beak", "feathers"], ) ``` ## Development To set up this plugin locally, first checkout the code. Then create a new virtual environment: ```bash cd llm-anthropic python3 -m venv venv source venv/bin/activate ``` Now install the dependencies and test dependencies: ```bash pip install -e . --group dev ``` To run the tests: ```bash pytest ``` Alternatively, if you have [uv](https://github.com/astral-sh/uv) you can run tests without first creating a virtual environment like this: ```bash uv run pytest uv run pytest -k test_tools ``` You can also run the `llm` command in a `uv` managed environment like this: ```bash uv run llm 'your prompt here' ``` To enable debug logs while running ([like this](https://github.com/simonw/llm-anthropic/issues/54#issuecomment-3536842831)), set this environment variable: ```bash export ANTHROPIC_LOG=debug ``` This project uses [pytest-recording](https://github.com/kiwicom/pytest-recording) to record Anthropic API responses for the tests, and [inline-snapshot](https://15r10nk.github.io/inline-snapshot/) for test assertions. If you add a new test that calls the API you can capture the API response like this: ```bash PYTEST_ANTHROPIC_API_KEY="$(llm keys get anthropic)" uv run pytest --record-mode once ``` You will need to have stored a valid Anthropic API key using this command first: ```bash llm keys set anthropic # Paste key here ``` To re-record all cassettes and update all inline snapshot assertions in one command: ```bash rm tests/cassettes/test_anthropic/*.yaml PYTEST_ANTHROPIC_API_KEY="$(llm keys get anthropic)" uv run pytest --record-mode all --inline-snapshot=fix ``` To re-record a single test: ```bash rm tests/cassettes/test_anthropic/test_thinking_prompt.yaml PYTEST_ANTHROPIC_API_KEY="$(llm keys get anthropic)" uv run pytest -k test_thinking_prompt --record-mode once --inline-snapshot=fix ``` ././@PaxHeader0000000000000000000000000000002600000000000010213 xustar0022 mtime=1785880938.0 llm_anthropic-0.26/llm_anthropic.egg-info/SOURCES.txt0000644000175100017510000000045115234460552022161 0ustar00runnerrunnerLICENSE README.md llm_anthropic.py pyproject.toml llm_anthropic.egg-info/PKG-INFO llm_anthropic.egg-info/SOURCES.txt llm_anthropic.egg-info/dependency_links.txt llm_anthropic.egg-info/entry_points.txt llm_anthropic.egg-info/requires.txt llm_anthropic.egg-info/top_level.txt tests/test_anthropic.py././@PaxHeader0000000000000000000000000000002600000000000010213 xustar0022 mtime=1785880938.0 llm_anthropic-0.26/llm_anthropic.egg-info/dependency_links.txt0000644000175100017510000000000115234460552024343 0ustar00runnerrunner ././@PaxHeader0000000000000000000000000000002600000000000010213 xustar0022 mtime=1785880938.0 llm_anthropic-0.26/llm_anthropic.egg-info/entry_points.txt0000644000175100017510000000004015234460552023565 0ustar00runnerrunner[llm] anthropic = llm_anthropic ././@PaxHeader0000000000000000000000000000002600000000000010213 xustar0022 mtime=1785880938.0 llm_anthropic-0.26/llm_anthropic.egg-info/requires.txt0000644000175100017510000000003415234460552022672 0ustar00runnerrunnerllm>=0.32 anthropic>=0.96.0 ././@PaxHeader0000000000000000000000000000002600000000000010213 xustar0022 mtime=1785880938.0 llm_anthropic-0.26/llm_anthropic.egg-info/top_level.txt0000644000175100017510000000001615234460552023024 0ustar00runnerrunnerllm_anthropic ././@PaxHeader0000000000000000000000000000002600000000000010213 xustar0022 mtime=1785880929.0 llm_anthropic-0.26/llm_anthropic.py0000644000175100017510000017215015234460541017161 0ustar00runnerrunnerfrom anthropic import Anthropic, AsyncAnthropic, transform_schema import enum import llm from llm.models import _partition_tools from llm.parts import ( AttachmentPart, Message, ReasoningPart, StreamEvent, TextPart, ToolCallPart, ToolResultPart, ) import json from typing import Any, Dict, Optional, List from urllib.parse import urlsplit from pydantic import Field, field_validator, model_validator DEFAULT_THINKING_TOKENS = 1024 DEFAULT_TEMPERATURE = 1.0 MCP_BETA = "mcp-client-2025-11-20" class ThinkingEffort(str, enum.Enum): LOW = "low" MEDIUM = "medium" HIGH = "high" XHIGH = "xhigh" MAX = "max" @llm.hookimpl def register_models(register): # https://docs.anthropic.com/claude/docs/models-overview register( ClaudeMessages("claude-3-opus-20240229"), AsyncClaudeMessages("claude-3-opus-20240229"), ) register( ClaudeMessages("claude-3-opus-latest"), AsyncClaudeMessages("claude-3-opus-latest"), aliases=("claude-3-opus",), ) register( ClaudeMessages("claude-3-sonnet-20240229"), AsyncClaudeMessages("claude-3-sonnet-20240229"), aliases=("claude-3-sonnet",), ) register( ClaudeMessages("claude-3-haiku-20240307"), AsyncClaudeMessages("claude-3-haiku-20240307"), aliases=("claude-3-haiku",), ) # 3.5 models register( ClaudeMessages( "claude-3-5-sonnet-20240620", supports_pdf=True, default_max_tokens=8192 ), AsyncClaudeMessages( "claude-3-5-sonnet-20240620", supports_pdf=True, default_max_tokens=8192 ), ) register( ClaudeMessages( "claude-3-5-sonnet-20241022", supports_pdf=True, supports_web_search=True, default_max_tokens=8192, ), AsyncClaudeMessages( "claude-3-5-sonnet-20241022", supports_pdf=True, supports_web_search=True, default_max_tokens=8192, ), ) register( ClaudeMessages( "claude-3-5-sonnet-latest", supports_pdf=True, supports_web_search=True, default_max_tokens=8192, ), AsyncClaudeMessages( "claude-3-5-sonnet-latest", supports_pdf=True, supports_web_search=True, default_max_tokens=8192, ), aliases=("claude-3.5-sonnet", "claude-3.5-sonnet-latest"), ) register( ClaudeMessages( "claude-3-5-haiku-latest", supports_web_search=True, default_max_tokens=8192 ), AsyncClaudeMessages( "claude-3-5-haiku-latest", supports_web_search=True, default_max_tokens=8192 ), aliases=("claude-3.5-haiku",), ) # 3.7 register( ClaudeMessages( "claude-3-7-sonnet-20250219", supports_pdf=True, supports_thinking=True, supports_web_search=True, default_max_tokens=8192, ), AsyncClaudeMessages( "claude-3-7-sonnet-20250219", supports_pdf=True, supports_thinking=True, supports_web_search=True, default_max_tokens=8192, ), ) register( ClaudeMessages( "claude-3-7-sonnet-latest", supports_pdf=True, supports_thinking=True, supports_web_search=True, default_max_tokens=8192, ), AsyncClaudeMessages( "claude-3-7-sonnet-latest", supports_pdf=True, supports_thinking=True, supports_web_search=True, default_max_tokens=8192, ), aliases=("claude-3.7-sonnet", "claude-3.7-sonnet-latest"), ) register( ClaudeMessages( "claude-opus-4-0", supports_pdf=True, supports_thinking=True, supports_web_search=True, default_max_tokens=32000, ), AsyncClaudeMessages( "claude-opus-4-0", supports_pdf=True, supports_thinking=True, supports_web_search=True, default_max_tokens=32000, ), aliases=("claude-4-opus",), ) register( ClaudeMessages( "claude-sonnet-4-0", supports_pdf=True, supports_thinking=True, supports_web_search=True, default_max_tokens=64000, ), AsyncClaudeMessages( "claude-sonnet-4-0", supports_pdf=True, supports_thinking=True, supports_web_search=True, default_max_tokens=64000, ), aliases=("claude-4-sonnet",), ) register( ClaudeMessages( "claude-opus-4-1-20250805", supports_pdf=True, supports_thinking=True, supports_web_search=True, use_structured_outputs=True, default_max_tokens=32000, ), AsyncClaudeMessages( "claude-opus-4-1-20250805", supports_pdf=True, supports_thinking=True, supports_web_search=True, use_structured_outputs=True, default_max_tokens=32000, ), aliases=("claude-opus-4.1",), ) # claude-sonnet-4-5 register( ClaudeMessages( "claude-sonnet-4-5", supports_pdf=True, supports_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=64000, ), AsyncClaudeMessages( "claude-sonnet-4-5", supports_pdf=True, supports_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=64000, ), aliases=("claude-sonnet-4.5",), ) # claude-haiku-4-5 register( ClaudeMessages( "claude-haiku-4-5-20251001", supports_pdf=True, supports_thinking=True, supports_web_search=True, supports_code_execution=True, default_max_tokens=64000, ), AsyncClaudeMessages( "claude-haiku-4-5-20251001", supports_pdf=True, supports_thinking=True, supports_web_search=True, supports_code_execution=True, default_max_tokens=64000, ), aliases=("claude-haiku-4.5",), ) # claude-opus-4-5 register( ClaudeMessages( "claude-opus-4-5-20251101", supports_pdf=True, supports_thinking=True, supports_thinking_effort=True, supports_web_search=True, supports_code_execution=True, default_max_tokens=64000, ), AsyncClaudeMessages( "claude-opus-4-5-20251101", supports_pdf=True, supports_thinking=True, supports_thinking_effort=True, supports_web_search=True, supports_code_execution=True, default_max_tokens=64000, ), aliases=("claude-opus-4.5",), ) # claude-opus-4-6 register( ClaudeMessages( "claude-opus-4-6", supports_pdf=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), AsyncClaudeMessages( "claude-opus-4-6", supports_pdf=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), aliases=("claude-opus-4.6",), ) # claude-sonnet-4-6 register( ClaudeMessages( "claude-sonnet-4-6", supports_pdf=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), AsyncClaudeMessages( "claude-sonnet-4-6", supports_pdf=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), aliases=("claude-sonnet-4.6",), ) # claude-opus-4-7 register( ClaudeMessages( "claude-opus-4-7", supports_pdf=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), AsyncClaudeMessages( "claude-opus-4-7", supports_pdf=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), aliases=("claude-opus-4.7",), ) # claude-opus-4-8 register( ClaudeMessages( "claude-opus-4-8", supports_pdf=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), AsyncClaudeMessages( "claude-opus-4-8", supports_pdf=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), aliases=("claude-opus-4.8",), ) # claude-fable-5 register( ClaudeMessages( "claude-fable-5", supports_pdf=True, thinks_by_default=True, always_thinks=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), AsyncClaudeMessages( "claude-fable-5", supports_pdf=True, thinks_by_default=True, always_thinks=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), aliases=("claude-fable-5",), ) # claude-sonnet-5 register( ClaudeMessages( "claude-sonnet-5", supports_pdf=True, thinks_by_default=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), AsyncClaudeMessages( "claude-sonnet-5", supports_pdf=True, thinks_by_default=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), aliases=("claude-sonnet-5",), ) # claude-opus-5 register( ClaudeMessages( "claude-opus-5", supports_pdf=True, thinks_by_default=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), AsyncClaudeMessages( "claude-opus-5", supports_pdf=True, thinks_by_default=True, supports_thinking=True, supports_thinking_effort=True, supports_adaptive_thinking=True, supports_web_search=True, supports_code_execution=True, use_structured_outputs=True, default_max_tokens=128000, ), aliases=("claude-opus-5",), ) class ClaudeOptions(llm.Options): max_tokens: int | None = Field( description="The maximum number of tokens to generate before stopping", default=None, ) temperature: float | None = Field( description="Amount of randomness injected into the response. Defaults to 1.0. Ranges from 0.0 to 1.0. Use temperature closer to 0.0 for analytical / multiple choice, and closer to 1.0 for creative and generative tasks. Note that even with temperature of 0.0, the results will not be fully deterministic.", default=None, ) top_p: float | None = Field( description="Use nucleus sampling. In nucleus sampling, we compute the cumulative distribution over all the options for each subsequent token in decreasing probability order and cut it off once it reaches a particular probability specified by top_p. You should either alter temperature or top_p, but not both. Recommended for advanced use cases only. You usually only need to use temperature.", default=None, ) top_k: int | None = Field( description="Only sample from the top K options for each subsequent token. Used to remove 'long tail' low probability responses. Recommended for advanced use cases only. You usually only need to use temperature.", default=None, ) user_id: str | None = Field( description="An external identifier for the user who is associated with the request", default=None, ) prefill: str | None = Field( description="A prefill to use for the response", default=None, ) hide_prefill: bool | None = Field( description="Do not repeat the prefill value at the start of the response", default=None, ) stop_sequences: list[str] | str | None = Field( description=( "Custom text sequences that will cause the model to stop generating - " "pass either a list of strings or a single string" ), default=None, ) cache: bool | None = Field( description="Use Anthropic prompt cache for any attachments or fragments", default=None, ) fast: bool | None = Field( description="Use fast mode for lower latency responses: https://platform.claude.com/docs/en/build-with-claude/fast-mode", default=None, ) @field_validator("stop_sequences") def validate_stop_sequences(cls, stop_sequences): error_msg = "stop_sequences must be a list of strings or a single string" if isinstance(stop_sequences, str): try: stop_sequences = json.loads(stop_sequences) if not isinstance(stop_sequences, list) or not all( isinstance(seq, str) for seq in stop_sequences ): raise ValueError(error_msg) return stop_sequences except json.JSONDecodeError: return [stop_sequences] elif isinstance(stop_sequences, list): if not all(isinstance(seq, str) for seq in stop_sequences): raise ValueError(error_msg) return stop_sequences else: raise ValueError(error_msg) @field_validator("temperature") @classmethod def validate_temperature(cls, temperature): if not (0.0 <= temperature <= 1.0): raise ValueError("temperature must be in range 0.0-1.0") return temperature @field_validator("top_p") @classmethod def validate_top_p(cls, top_p): if top_p is not None and not (0.0 <= top_p <= 1.0): raise ValueError("top_p must be in range 0.0-1.0") return top_p @field_validator("top_k") @classmethod def validate_top_k(cls, top_k): if top_k is not None and top_k <= 0: raise ValueError("top_k must be a positive integer") return top_k @model_validator(mode="after") def validate_temperature_top_p(self): if self.temperature != 1.0 and self.top_p is not None: raise ValueError("Only one of temperature and top_p can be set") return self class ClaudeOptionsWithThinking(ClaudeOptions): thinking: bool | None = Field( description=( "Enable thinking mode. Claude 5 models think by default - " "set to false to disable thinking on models that allow it" ), default=None, ) class ClaudeOptionsWithThinkingEffort(ClaudeOptionsWithThinking): thinking_effort: ThinkingEffort | None = Field( description="Level of thinking effort to apply: low, medium, high, xhigh or max", default=None, ) def _validate_max_uses(max_uses): if max_uses is not None and ( isinstance(max_uses, bool) or not isinstance(max_uses, int) or max_uses < 1 ): raise ValueError("max_uses must be a positive integer") def _validate_domain_filters(allowed_domains, blocked_domains): if allowed_domains is not None and blocked_domains is not None: raise ValueError("Cannot specify both allowed_domains and blocked_domains") for name, domains in ( ("allowed_domains", allowed_domains), ("blocked_domains", blocked_domains), ): if domains is None: continue if not isinstance(domains, list) or not all( isinstance(domain, str) and domain for domain in domains ): raise ValueError(f"{name} must be a list of non-empty strings") class WebSearch(llm.ServerSideTool): """Search the web using Anthropic's server-side web search tool. On Claude 4.6 and later models this uses ``web_search_20260318`` with dynamic content filtering; older models use ``web_search_20250305``. """ name = "web_search" def __init__( self, max_uses: Optional[int] = None, allowed_domains: Optional[List[str]] = None, blocked_domains: Optional[List[str]] = None, user_location: Optional[dict] = None, ): super().__init__() _validate_max_uses(max_uses) _validate_domain_filters(allowed_domains, blocked_domains) if user_location is not None: if not isinstance(user_location, dict): raise ValueError("user_location must be a dictionary") allowed_keys = {"type", "city", "region", "country", "timezone"} invalid_keys = set(user_location.keys()) - allowed_keys if invalid_keys: raise ValueError( f"user_location contains invalid keys: {invalid_keys}. " f"Allowed keys: {allowed_keys}" ) user_location = dict(user_location) user_location.setdefault("type", "approximate") if user_location["type"] != "approximate": raise ValueError("user_location type must be approximate") self.max_uses = max_uses self.allowed_domains = allowed_domains self.blocked_domains = blocked_domains self.user_location = user_location def tool_spec(self, model): modern = getattr(model, "supports_adaptive_thinking", False) spec = { "type": "web_search_20260318" if modern else "web_search_20250305", "name": "web_search", } if self.max_uses is not None: spec["max_uses"] = self.max_uses if self.allowed_domains is not None: spec["allowed_domains"] = list(self.allowed_domains) if self.blocked_domains is not None: spec["blocked_domains"] = list(self.blocked_domains) if self.user_location is not None: spec["user_location"] = dict(self.user_location) return spec class WebFetch(llm.ServerSideTool): """Fetch the full contents of a URL using Anthropic's server-side web fetch tool. Claude can only fetch URLs that already appear in the conversation - provided by the user or returned by a previous web search or fetch. On Claude 4.6 and later models this uses ``web_fetch_20260318`` with dynamic content filtering; older models use ``web_fetch_20250910``. """ name = "web_fetch" def __init__( self, max_uses: Optional[int] = None, allowed_domains: Optional[List[str]] = None, blocked_domains: Optional[List[str]] = None, citations: bool = False, max_content_tokens: Optional[int] = None, use_cache: Optional[bool] = None, ): super().__init__() _validate_max_uses(max_uses) _validate_domain_filters(allowed_domains, blocked_domains) if not isinstance(citations, bool): raise ValueError("citations must be a boolean") if max_content_tokens is not None and ( isinstance(max_content_tokens, bool) or not isinstance(max_content_tokens, int) or max_content_tokens < 1 ): raise ValueError("max_content_tokens must be a positive integer") if use_cache is not None and not isinstance(use_cache, bool): raise ValueError("use_cache must be a boolean") self.max_uses = max_uses self.allowed_domains = allowed_domains self.blocked_domains = blocked_domains self.citations = citations self.max_content_tokens = max_content_tokens self.use_cache = use_cache def tool_spec(self, model): modern = getattr(model, "supports_adaptive_thinking", False) if self.use_cache is not None and not modern: raise ValueError( f"use_cache is not supported by model {model.model_id} - " "it requires a Claude 4.6 or later model" ) spec = { "type": "web_fetch_20260318" if modern else "web_fetch_20250910", "name": "web_fetch", } if self.max_uses is not None: spec["max_uses"] = self.max_uses if self.allowed_domains is not None: spec["allowed_domains"] = list(self.allowed_domains) if self.blocked_domains is not None: spec["blocked_domains"] = list(self.blocked_domains) if self.citations: spec["citations"] = {"enabled": True} if self.max_content_tokens is not None: spec["max_content_tokens"] = self.max_content_tokens if self.use_cache is not None: spec["use_cache"] = self.use_cache return spec class AnthropicMCP(llm.ServerSideTool): """Call tools on a remote MCP server using Anthropic's MCP connector. Anthropic connects to the MCP server from their own infrastructure - the server must be reachable over HTTPS. Uses the ``mcp-client-2025-11-20`` beta. Only MCP tool calls are supported (not resources or prompts). """ name = "mcp" def __init__( self, url: str, name: Optional[str] = None, authorization_token: Optional[str] = None, allowed_tools: Optional[List[str]] = None, ): super().__init__() if not isinstance(url, str) or not url: raise ValueError("url must be a non-empty string") if not url.startswith("https://"): raise ValueError("url must start with https://") if name is not None and (not isinstance(name, str) or not name): raise ValueError("name must be a non-empty string") if authorization_token is not None and not isinstance(authorization_token, str): raise ValueError("authorization_token must be a string") if allowed_tools is not None: if not isinstance(allowed_tools, list) or not all( isinstance(tool_name, str) and tool_name for tool_name in allowed_tools ): raise ValueError("allowed_tools must be a list of non-empty strings") self.url = url self.server_name = name or urlsplit(url).hostname self.authorization_token = authorization_token self.allowed_tools = allowed_tools def tool_spec(self, model): spec = {"type": "mcp_toolset", "mcp_server_name": self.server_name} if self.allowed_tools is not None: spec["default_config"] = {"enabled": False} spec["configs"] = { tool_name: {"enabled": True} for tool_name in self.allowed_tools } return spec def prepare_request(self, model, kwargs): server = {"type": "url", "url": self.url, "name": self.server_name} if self.authorization_token is not None: server["authorization_token"] = self.authorization_token servers = kwargs.setdefault("mcp_servers", []) if not any(existing.get("name") == self.server_name for existing in servers): servers.append(server) betas = kwargs.setdefault("betas", []) if MCP_BETA not in betas: betas.append(MCP_BETA) class CodeExecution(llm.ServerSideTool): """Run Python and bash code in Anthropic's sandboxed server-side execution container. Pass an existing container ID as ``container`` to reuse the files and state from a previous response - the container ID is available in the logged response JSON. """ name = "code_execution" def __init__(self, container: Optional[str] = None): super().__init__() if container is not None and not isinstance(container, str): raise ValueError("container must be a string container ID") self.container = container def tool_spec(self, model): return {"type": "code_execution_20260521", "name": "code_execution"} def prepare_request(self, model, kwargs): if self.container is not None: kwargs["container"] = self.container def source_for_attachment(attachment): if attachment.url: return { "type": "url", "url": attachment.url, } else: return { "data": attachment.base64_content(), "media_type": attachment.resolve_type(), "type": "base64", } class _Shared: needs_key = "anthropic" key_env_var = "ANTHROPIC_API_KEY" can_stream = True base_url = None supports_thinking = False supports_thinking_effort = False supports_adaptive_thinking = False supports_schema = True supports_tools = True supports_web_search = False supports_code_execution = False thinks_by_default = False always_thinks = False default_max_tokens = 4096 class Options(ClaudeOptions): ... def __init__( self, model_id, claude_model_id=None, supports_images=True, supports_pdf=False, supports_thinking=False, supports_thinking_effort=False, supports_adaptive_thinking=False, supports_web_search=False, supports_code_execution=False, thinks_by_default=False, always_thinks=False, use_structured_outputs=False, default_max_tokens=None, base_url=None, ): self.model_id = "anthropic/" + model_id self.claude_model_id = claude_model_id or model_id self.base_url = base_url self.use_structured_outputs = use_structured_outputs self.attachment_types = set() if supports_images: self.attachment_types.update( { "image/png", "image/jpeg", "image/webp", "image/gif", } ) if supports_pdf: self.attachment_types.add("application/pdf") if supports_thinking: self.supports_thinking = True self.Options = ClaudeOptionsWithThinking if supports_thinking_effort: self.supports_thinking_effort = True self.Options = ClaudeOptionsWithThinkingEffort if supports_adaptive_thinking: self.supports_adaptive_thinking = True if default_max_tokens is not None: self.default_max_tokens = default_max_tokens self.supports_web_search = supports_web_search self.supports_code_execution = supports_code_execution self.thinks_by_default = thinks_by_default self.always_thinks = always_thinks @property def supported_server_side_tools(self): tools = [] if self.supports_web_search: tools += [WebSearch, WebFetch, AnthropicMCP] if self.supports_code_execution: tools.append(CodeExecution) return tuple(tools) def prefill_text(self, prompt): if prompt.options.prefill and not prompt.options.hide_prefill: return prompt.options.prefill return "" def _server_tool_result_event(self, block_type, block) -> StreamEvent: """Build a tool_result StreamEvent from a server tool result block. web_search_tool_result content is a list of result blocks; web_fetch_tool_result content is a single result object. """ content = getattr(block, "content", None) if isinstance(content, list): result_text = ( json.dumps( [b if isinstance(b, dict) else b.model_dump() for b in content] ) if content else "" ) elif content is None: result_text = "" else: result_text = json.dumps( content if isinstance(content, dict) else content.model_dump() ) return StreamEvent( type="tool_result", chunk=result_text, tool_call_id=getattr(block, "tool_use_id", None), server_executed=True, tool_name=block_type.removesuffix("_tool_result"), ) def _apply_container(self, message_dict, container): """Store the code execution container on the response JSON. Streaming needs this patched in from the message_delta event - the SDK's get_final_message() accumulator drops it - and in both modes the datetime expires_at must become a JSON-safe string. """ if container is None: container = message_dict.get("container") if container is None: return if hasattr(container, "model_dump"): container = container.model_dump(mode="json") else: container = { key: value.isoformat() if hasattr(value, "isoformat") else value for key, value in container.items() } message_dict["container"] = container def _model_dump_suppress_warnings(self, message): """ Call model_dump() on a message while suppressing Pydantic serialization warnings. When using dynamically created Pydantic models with the SDK's stream() helper, the returned ParsedBetaMessage has strict type annotations that don't match our dynamic models, causing harmless serialization warnings. This suppresses those warnings while still producing correct output. """ import warnings with warnings.catch_warnings(): warnings.filterwarnings("ignore", category=UserWarning, module="pydantic") return message.model_dump() # --- messages= support ------------------------------------------------- # # This plugin consumes prompt.messages (the canonical list[Message] # that llm synthesizes from legacy inputs when messages= wasn't # explicitly passed). Each Message + its Parts is translated into # Anthropic content blocks; adjacent user-side messages (role="user" # or role="tool") are merged because Anthropic requires alternating # user/assistant turns. def _part_to_block(self, part) -> Optional[Dict[str, Any]]: """Translate one llm Part into an Anthropic content block.""" pm = getattr(part, "provider_metadata", None) or {} anthropic_pm = pm.get("anthropic", {}) if isinstance(pm, dict) else {} if isinstance(part, TextPart): block: Dict[str, Any] = {"type": "text", "text": part.text} return block if isinstance(part, ReasoningPart): block = {"type": "thinking", "thinking": part.text} # Anthropic signed-thinking requires the signature echoed back. sig = ( anthropic_pm.get("signature") if isinstance(anthropic_pm, dict) else None ) if sig: block["signature"] = sig return block if isinstance(part, ToolCallPart): mcp_server_name = ( anthropic_pm.get("mcp_server_name") if isinstance(anthropic_pm, dict) else None ) if part.server_executed and ( mcp_server_name or (part.tool_call_id or "").startswith("mcptoolu") ): # MCP connector calls replay as mcp_tool_use blocks; the # API requires the server_name field to be echoed back. block = { "type": "mcp_tool_use", "id": part.tool_call_id, "name": part.name, "input": part.arguments, } if mcp_server_name: block["server_name"] = mcp_server_name return block return { "type": "server_tool_use" if part.server_executed else "tool_use", "id": part.tool_call_id, "name": part.name, "input": part.arguments, } if isinstance(part, ToolResultPart): if part.server_executed: # Reconstruct the provider result block that arrived in the # assistant turn (e.g. web_fetch_tool_result) - the API # rejects plain tool_result blocks in assistant messages. try: content = json.loads(part.output) if part.output else None except ValueError: content = part.output block = { "type": part.name + "_tool_result", "tool_use_id": part.tool_call_id, } if content is not None: block["content"] = content return block return { "type": "tool_result", "tool_use_id": part.tool_call_id, "content": part.output, } if isinstance(part, AttachmentPart) and part.attachment is not None: attachment = part.attachment attachment_type = ( "document" if attachment.resolve_type() == "application/pdf" else "image" ) return { "type": attachment_type, "source": source_for_attachment(attachment), } return None def _message_to_blocks(self, message: Message) -> List[Dict[str, Any]]: blocks: List[Dict[str, Any]] = [] for part in message.parts: block = self._part_to_block(part) if block is not None: blocks.append(block) if message.role == "assistant": filtered_blocks: List[Dict[str, Any]] = [] seen_tool_use = False for block in blocks: block_type = block.get("type") if seen_tool_use and block_type == "text" and block.get("text") == " ": # The sync streaming path yields a display-only space # after tool calls so chained text does not run together. # Anthropic rejects assistant history that places text # after tool_use instead of immediately before tool_result. continue filtered_blocks.append(block) if block_type == "tool_use": seen_tool_use = True blocks = filtered_blocks return blocks def _append_message(self, out: List[Dict[str, Any]], message: Message) -> None: """Append an Anthropic-shaped message, merging with the previous one if both would be user-side turns (tool_result + text in the same user message is the required shape for tool follow-ups).""" if message.role == "system": return # system lives on the top-level kwargs["system"] field blocks = self._message_to_blocks(message) if not blocks: return # Anthropic: tool messages from llm become user messages with # tool_result blocks; assistant stays assistant. anthropic_role = "assistant" if message.role == "assistant" else "user" if out and out[-1]["role"] == anthropic_role and anthropic_role == "user": out[-1]["content"].extend(blocks) else: out.append({"role": anthropic_role, "content": blocks}) def _append_prev_response_output( self, out: List[Dict[str, Any]], prev_response ) -> None: """Add the assistant turn from a previous Response. Mirrors the flat text+tool_calls pattern used by the OpenAI plugin.""" assistant_content: List[Dict[str, Any]] = [] text_content = prev_response.text_or_raise() if text_content: assistant_content.append({"type": "text", "text": text_content}) for tool_call in prev_response.tool_calls_or_raise(): assistant_content.append( { "type": "tool_use", "id": tool_call.tool_call_id, "name": tool_call.name, "input": tool_call.arguments, } ) if assistant_content: out.append({"role": "assistant", "content": assistant_content}) def build_messages(self, prompt, conversation) -> list[dict]: messages: List[Dict[str, Any]] = [] # Current turn — iterate prompt.messages (auto-synthesized from # legacy inputs if messages= was not explicitly passed). In llm # 0.32+ conversation and chain paths pre-bake the full input chain # here, so also walking conversation.responses would duplicate # prior turns and break tool-result ordering. for message in prompt.messages: self._append_message(messages, message) # Cache control: apply to the last content block of the final # user-side turn, matching the pre-upgrade behavior. if prompt.options.cache and messages: last_message = messages[-1] if ( isinstance(last_message.get("content"), list) and last_message["content"] ): last_message["content"][-1]["cache_control"] = {"type": "ephemeral"} # Prefill — append an assistant turn the model will continue from. if prompt.options.prefill: if self.supports_adaptive_thinking: raise ValueError( f"Prefilling assistant messages is not supported by {self.claude_model_id}. " f"Use structured outputs or system prompt instructions instead." ) messages.append( { "role": "assistant", "content": [{"type": "text", "text": prompt.options.prefill}], } ) return messages def _extract_system(self, prompt) -> Optional[str]: """Pull the system prompt from prompt.messages or prompt.system. ``prompt.system`` already composes ``_system`` + ``system_fragments``; if messages= was passed explicitly and it contains a system-role message, fall back to reading that. """ if prompt.system: return prompt.system for message in prompt.messages: if message.role == "system": texts = [p.text for p in message.parts if isinstance(p, TextPart)] if texts: return "\n\n".join(texts) return None def build_kwargs(self, prompt, conversation): if prompt.schema and prompt.tools: raise ValueError( "llm-anthropic does not yet support using both schema and tools in the same prompt" ) kwargs = { "model": self.claude_model_id, "messages": self.build_messages(prompt, conversation), } if prompt.options.user_id: kwargs["metadata"] = {"user_id": prompt.options.user_id} if prompt.options.top_p: kwargs["top_p"] = prompt.options.top_p else: kwargs["temperature"] = ( prompt.options.temperature if prompt.options.temperature is not None else DEFAULT_TEMPERATURE ) if prompt.options.top_k: kwargs["top_k"] = prompt.options.top_k system = self._extract_system(prompt) if system: kwargs["system"] = system if prompt.options.stop_sequences: kwargs["stop_sequences"] = prompt.options.stop_sequences thinking_effort_enabled = ( self.supports_thinking_effort and prompt.options.thinking_effort ) # Thinking: Claude 5 models think by default (adaptive mode); # older models only think when it is explicitly requested. if self.supports_thinking: hide_reasoning = getattr(prompt, "hide_reasoning", False) if prompt.options.thinking is False: if self.always_thinks: raise ValueError( f"Thinking cannot be disabled for model {self.model_id}" ) kwargs["thinking"] = {"type": "disabled"} elif prompt.options.thinking or thinking_effort_enabled: if self.supports_adaptive_thinking or thinking_effort_enabled: kwargs["thinking"] = {"type": "adaptive"} else: # Pre-4.6 models: enabled with default budget kwargs["thinking"] = { "type": "enabled", "budget_tokens": DEFAULT_THINKING_TOKENS, } elif self.thinks_by_default and hide_reasoning: # No thinking option set, but the model will think anyway - # send the param explicitly so display can be omitted below kwargs["thinking"] = {"type": "adaptive"} if ( hide_reasoning and "thinking" in kwargs and kwargs["thinking"]["type"] != "disabled" ): # -R / hide_reasoning=True asks the API to leave the # thinking trace out of the response entirely kwargs["thinking"]["display"] = "omitted" # Handle effort in output_config if thinking_effort_enabled: kwargs.setdefault("output_config", {})[ "effort" ] = prompt.options.thinking_effort.value max_tokens = self.default_max_tokens if prompt.options.max_tokens is not None: max_tokens = prompt.options.max_tokens kwargs["max_tokens"] = max_tokens # Determine which beta headers to use betas = [] # Effort beta: only for pre-GA models (e.g., Opus 4.5) if ( "output_config" in kwargs and "effort" in kwargs.get("output_config", {}) and not self.supports_adaptive_thinking ): betas.append("effort-2025-11-24") # 128K output beta: not needed for 4.6 models if max_tokens > 64000 and not self.supports_adaptive_thinking: betas.append("output-128k-2025-02-19") if "thinking" in kwargs: kwargs["extra_body"] = {"thinking": kwargs.pop("thinking")} # Check if we should use new structured outputs use_structured_outputs = prompt.schema and self.use_structured_outputs if use_structured_outputs: kwargs.setdefault("output_config", {})["format"] = { "type": "json_schema", "schema": transform_schema(prompt.schema), } # Fast mode for lower latency responses if prompt.options.fast: kwargs["speed"] = "fast" betas.append("fast-mode-2026-02-01") if betas: kwargs["betas"] = betas tools = [] if prompt.schema and not use_structured_outputs: # Fall back to tools workaround for models that don't support structured outputs tools.append( { "name": "output_structured_data", "input_schema": prompt.schema, } ) kwargs["tool_choice"] = {"type": "tool", "name": "output_structured_data"} server_side_tools = [] if prompt.tools: function_tools, server_side_tools = _partition_tools(self, prompt.tools) tools.extend(tool.tool_spec(self) for tool in server_side_tools) tools.extend( [ { "name": tool.name, "description": tool.description or "", "input_schema": tool.input_schema, } for tool in function_tools ] ) if tools: kwargs["tools"] = tools for tool in server_side_tools: tool.prepare_request(self, kwargs) return kwargs def set_usage(self, response): usage = response.response_json.pop("usage") input_tokens = usage.pop("input_tokens") output_tokens = usage.pop("output_tokens") # Only include usage details if prompt caching was on or web search was used details = None if response.prompt.options.cache or usage.get("server_tool_use"): details = usage response.set_usage(input=input_tokens, output=output_tokens, details=details) def add_tool_usage(self, response, last_message) -> bool: tool_uses = [ item for item in last_message["content"] if item["type"] == "tool_use" ] for tool_use in tool_uses: response.add_tool_call( llm.ToolCall( tool_call_id=tool_use["id"], name=tool_use["name"], arguments=tool_use["input"], ) ) return bool(tool_uses) def __str__(self): return "Anthropic Messages: {}".format(self.model_id) class ClaudeMessages(_Shared, llm.KeyModel): def execute(self, prompt, stream, response, conversation, key): client = Anthropic(api_key=self.get_key(key), base_url=self.base_url) kwargs = self.build_kwargs(prompt, conversation) prefill_text = self.prefill_text(prompt) if "betas" in kwargs: messages_client = client.beta.messages else: messages_client = client.messages if stream: with messages_client.stream(**kwargs) as stream_obj: current_block_id = None current_block_name = None is_server_tool = False container = None if prefill_text: yield StreamEvent(type="text", chunk=prefill_text) for chunk in stream_obj: if chunk.type == "content_block_start": block = chunk.content_block block_type = getattr(block, "type", None) current_block_id = getattr(block, "id", None) current_block_name = getattr(block, "name", None) is_server_tool = block_type in ( "server_tool_use", "mcp_tool_use", ) or (block_type or "").endswith("_tool_result") if block_type in ( "tool_use", "server_tool_use", "mcp_tool_use", ): yield StreamEvent( type="tool_call_name", chunk=current_block_name or "", tool_call_id=current_block_id, server_executed=(block_type != "tool_use"), provider_metadata=( { "anthropic": { "mcp_server_name": getattr( block, "server_name", None ) } } if block_type == "mcp_tool_use" else None ), ) elif block_type and block_type.endswith("_tool_result"): # Content is available inline on content_block_start yield self._server_tool_result_event(block_type, block) elif chunk.type == "content_block_delta": delta = chunk.delta delta_type = getattr(delta, "type", None) if delta_type == "thinking_delta": yield StreamEvent(type="reasoning", chunk=delta.thinking) elif delta_type == "signature_delta": yield StreamEvent( type="reasoning", chunk="", provider_metadata={ "anthropic": {"signature": delta.signature} }, ) elif delta_type == "text_delta": yield StreamEvent(type="text", chunk=delta.text) elif delta_type == "input_json_delta": yield StreamEvent( type="tool_call_args", chunk=delta.partial_json, tool_call_id=current_block_id, server_executed=is_server_tool, ) elif chunk.type == "message_delta": chunk_container = getattr(chunk, "container", None) or getattr( getattr(chunk, "delta", None), "container", None ) if chunk_container is not None: container = chunk_container # This records usage and other data: last_message = self._model_dump_suppress_warnings( stream_obj.get_final_message() ) self._apply_container(last_message, container) response.response_json = last_message if self.add_tool_usage(response, last_message): # Avoid "can have dragons.Now that I " bug yield StreamEvent(type="text", chunk=" ") else: completion = messages_client.create(**kwargs) for item in completion.content: item_type = getattr(item, "type", None) if item_type == "thinking": signature = getattr(item, "signature", None) yield StreamEvent( type="reasoning", chunk=item.thinking, provider_metadata=( {"anthropic": {"signature": signature}} if signature else None ), ) elif item_type == "text": text = (prefill_text + item.text) if prefill_text else item.text prefill_text = "" # Only prepend once yield StreamEvent(type="text", chunk=text) elif item_type in ("tool_use", "server_tool_use", "mcp_tool_use"): server_executed = item_type != "tool_use" yield StreamEvent( type="tool_call_name", chunk=item.name, tool_call_id=item.id, server_executed=server_executed, provider_metadata=( { "anthropic": { "mcp_server_name": getattr( item, "server_name", None ) } } if item_type == "mcp_tool_use" else None ), ) yield StreamEvent( type="tool_call_args", chunk=json.dumps(item.input), tool_call_id=item.id, server_executed=server_executed, ) elif item_type and item_type.endswith("_tool_result"): yield self._server_tool_result_event(item_type, item) response.response_json = completion.model_dump() self._apply_container(response.response_json, None) self.add_tool_usage(response, response.response_json) self.set_usage(response) class AsyncClaudeMessages(_Shared, llm.AsyncKeyModel): async def execute(self, prompt, stream, response, conversation, key): client = AsyncAnthropic(api_key=self.get_key(key), base_url=self.base_url) kwargs = self.build_kwargs(prompt, conversation) if "betas" in kwargs: messages_client = client.beta.messages else: messages_client = client.messages prefill_text = self.prefill_text(prompt) if stream: async with messages_client.stream(**kwargs) as stream_obj: current_block_id = None current_block_name = None is_server_tool = False container = None if prefill_text: yield StreamEvent(type="text", chunk=prefill_text) async for chunk in stream_obj: if chunk.type == "content_block_start": block = chunk.content_block block_type = getattr(block, "type", None) current_block_id = getattr(block, "id", None) current_block_name = getattr(block, "name", None) is_server_tool = block_type in ( "server_tool_use", "mcp_tool_use", ) or (block_type or "").endswith("_tool_result") if block_type in ( "tool_use", "server_tool_use", "mcp_tool_use", ): yield StreamEvent( type="tool_call_name", chunk=current_block_name or "", tool_call_id=current_block_id, server_executed=(block_type != "tool_use"), provider_metadata=( { "anthropic": { "mcp_server_name": getattr( block, "server_name", None ) } } if block_type == "mcp_tool_use" else None ), ) elif block_type and block_type.endswith("_tool_result"): yield self._server_tool_result_event(block_type, block) elif chunk.type == "content_block_delta": delta = chunk.delta delta_type = getattr(delta, "type", None) if delta_type == "thinking_delta": yield StreamEvent(type="reasoning", chunk=delta.thinking) elif delta_type == "signature_delta": yield StreamEvent( type="reasoning", chunk="", provider_metadata={ "anthropic": {"signature": delta.signature} }, ) elif delta_type == "text_delta": yield StreamEvent(type="text", chunk=delta.text) elif delta_type == "input_json_delta": yield StreamEvent( type="tool_call_args", chunk=delta.partial_json, tool_call_id=current_block_id, server_executed=is_server_tool, ) elif chunk.type == "message_delta": chunk_container = getattr(chunk, "container", None) or getattr( getattr(chunk, "delta", None), "container", None ) if chunk_container is not None: container = chunk_container response.response_json = self._model_dump_suppress_warnings( await stream_obj.get_final_message() ) self._apply_container(response.response_json, container) self.add_tool_usage(response, response.response_json) else: completion = await messages_client.create(**kwargs) for item in completion.content: item_type = getattr(item, "type", None) if item_type == "thinking": signature = getattr(item, "signature", None) yield StreamEvent( type="reasoning", chunk=item.thinking, provider_metadata=( {"anthropic": {"signature": signature}} if signature else None ), ) elif item_type == "text": text = (prefill_text + item.text) if prefill_text else item.text prefill_text = "" yield StreamEvent(type="text", chunk=text) elif item_type in ("tool_use", "server_tool_use", "mcp_tool_use"): server_executed = item_type != "tool_use" yield StreamEvent( type="tool_call_name", chunk=item.name, tool_call_id=item.id, server_executed=server_executed, provider_metadata=( { "anthropic": { "mcp_server_name": getattr( item, "server_name", None ) } } if item_type == "mcp_tool_use" else None ), ) yield StreamEvent( type="tool_call_args", chunk=json.dumps(item.input), tool_call_id=item.id, server_executed=server_executed, ) elif item_type and item_type.endswith("_tool_result"): yield self._server_tool_result_event(item_type, item) response.response_json = completion.model_dump() self._apply_container(response.response_json, None) self.add_tool_usage(response, response.response_json) self.set_usage(response) ././@PaxHeader0000000000000000000000000000002600000000000010213 xustar0022 mtime=1785880929.0 llm_anthropic-0.26/pyproject.toml0000644000175100017510000000152615234460541016666 0ustar00runnerrunner[project] name = "llm-anthropic" version = "0.26" description = "LLM access to models by Anthropic, including the Claude series" readme = "README.md" authors = [{name = "Simon Willison"}] license = "Apache-2.0" requires-python = ">=3.10" classifiers = [] dependencies = [ "llm>=0.32", "anthropic>=0.96.0", ] [project.urls] Homepage = "https://github.com/simonw/llm-anthropic" Changelog = "https://github.com/simonw/llm-anthropic/releases" Issues = "https://github.com/simonw/llm-anthropic/issues" CI = "https://github.com/simonw/llm-anthropic/actions" [project.entry-points.llm] anthropic = "llm_anthropic" [dependency-groups] dev = ["pytest", "pytest-recording", "pytest-asyncio", "cogapp", "inline-snapshot[black]"] [tool.uv] package = true [tool.pytest.ini_options] asyncio_mode = "strict" asyncio_default_fixture_loop_scope = "function" ././@PaxHeader0000000000000000000000000000003400000000000010212 xustar0028 mtime=1785880938.9597065 llm_anthropic-0.26/setup.cfg0000644000175100017510000000004615234460553015572 0ustar00runnerrunner[egg_info] tag_build = tag_date = 0 ././@PaxHeader0000000000000000000000000000003300000000000010211 xustar0027 mtime=1785880938.959083 llm_anthropic-0.26/tests/0000755000175100017510000000000015234460553015113 5ustar00runnerrunner././@PaxHeader0000000000000000000000000000002600000000000010213 xustar0022 mtime=1785880929.0 llm_anthropic-0.26/tests/test_anthropic.py0000644000175100017510000015267515234460541020530 0ustar00runnerrunnerimport json import llm import os import pytest from inline_snapshot import snapshot from pydantic import BaseModel TINY_PNG = ( b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\xa6\x00\x00\x01\x1a" b"\x02\x03\x00\x00\x00\xe6\x99\xc4^\x00\x00\x00\tPLTE\xff\xff\xff" b"\x00\xff\x00\xfe\x01\x00\x12t\x01J\x00\x00\x00GIDATx\xda\xed\xd81\x11" b"\x000\x08\xc0\xc0.]\xea\xaf&Q\x89\x04V\xe0>\xf3+\xc8\x91Z\xf4\xa2\x08EQ\x14E" b"Q\x14EQ\x14EQ\xd4B\x91$I3\xbb\xbf\x08EQ\x14EQ\x14EQ\x14E\xd1\xa5" b"\xd4\x17\x91\xc6\x95\x05\x15\x0f\x9f\xc5\t\x9f\xa4\x00\x00\x00\x00IEND\xaeB`" b"\x82" ) ANTHROPIC_API_KEY = os.environ.get("PYTEST_ANTHROPIC_API_KEY", None) or "sk-..." FIXED_TEST_VERSION = "0.32a0" def fixed_version_tool(): def fixed_version() -> str: return FIXED_TEST_VERSION return llm.Tool.function( fixed_version, name="fixed_version", description="Return a fixed test version string", ) @pytest.mark.vcr def test_prompt(): model = llm.get_model("claude-sonnet-4.5") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt("Two names for a pet pelican, be brief") assert str(response) == snapshot("- Captain\n- Scoop") response_dict = dict(response.response_json) response_dict.pop("id") # differs between requests assert response_dict == snapshot( { "container": None, "content": [ { "citations": None, "parsed_output": None, "text": "- Captain\n- Scoop", "type": "text", } ], "model": "claude-sonnet-4-5-20250929", "role": "assistant", "stop_details": None, "stop_reason": "end_turn", "stop_sequence": None, "type": "message", } ) assert response.input_tokens == snapshot(17) assert response.output_tokens == snapshot(10) assert response.token_details is None @pytest.mark.vcr @pytest.mark.asyncio async def test_async_prompt(): model = llm.get_async_model("claude-sonnet-4.5") model.key = model.key or ANTHROPIC_API_KEY # don't override existing key conversation = model.conversation() response = await conversation.prompt("Two names for a pet pelican, be brief") assert await response.text() == snapshot("- Captain\n- Scoop") response_dict = dict(response.response_json) response_dict.pop("id") # differs between requests assert response_dict == snapshot( { "container": None, "content": [ { "citations": None, "parsed_output": None, "text": "- Captain\n- Scoop", "type": "text", } ], "model": "claude-sonnet-4-5-20250929", "role": "assistant", "stop_details": None, "stop_reason": "end_turn", "stop_sequence": None, "type": "message", } ) assert response.input_tokens == snapshot(17) assert response.output_tokens == snapshot(10) assert response.token_details is None response2 = await conversation.prompt("in french") assert await response2.text() == snapshot("- Capitaine\n- Bec (beak)") @pytest.mark.vcr def test_image_prompt(): model = llm.get_model("claude-sonnet-4.5") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt( "Describe image in three words", attachments=[llm.Attachment(content=TINY_PNG)], ) assert str(response) == snapshot("Red square, green square.") response_dict = response.response_json response_dict.pop("id") # differs between requests assert response_dict == snapshot( { "container": None, "content": [ { "citations": None, "parsed_output": None, "text": "Red square, green square.", "type": "text", } ], "model": "claude-sonnet-4-5-20250929", "role": "assistant", "stop_details": None, "stop_reason": "end_turn", "stop_sequence": None, "type": "message", } ) assert response.input_tokens == snapshot(83) assert response.output_tokens == snapshot(9) assert response.token_details is None @pytest.mark.vcr def test_image_with_no_prompt(): model = llm.get_model("claude-sonnet-4.5") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt( prompt=None, attachments=[llm.Attachment(content=TINY_PNG)], ) assert str(response) == snapshot( "I need to describe what I see in this image.\n\n" "The image shows two solid colored rectangles arranged vertically on a white background:\n\n" "1. **Top rectangle**: A bright red rectangle positioned in the upper portion of the image\n" "2. **Bottom rectangle**: A bright green (lime green) rectangle positioned in the lower portion of the image\n\n" "Both rectangles appear to be roughly the same size and shape (horizontal rectangles/landscape orientation), " "and they are separated by white space between them." ) @pytest.mark.vcr def test_url_prompt(): model = llm.get_model("claude-sonnet-4.5") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt( prompt="describe image", attachments=[ llm.Attachment( url="https://static.simonwillison.net/static/2024/pelican.jpg" ) ], ) assert str(response) == snapshot( "This image shows a **brown pelican** perched on rocky terrain at what appears " "to be a marina or harbor. The pelican is captured in profile, displaying its " "distinctive features:\n\n" "- **Long, prominent bill** with the characteristic pelican pouch\n" "- **White head and neck** with darker gray-brown plumage on its body and wings\n" "- **Sturdy build** with detailed feather texture visible in the wings\n\n" "The background shows several **boats docked in a marina**, slightly out of " "focus, creating a typical coastal or waterfront setting. The lighting suggests " "this photo was taken during daytime, with bright natural light that creates " "a slight halo effect around the bird's head.\n\n" "The pelican appears calm and at rest, which is common behavior for these " "seabirds in harbor areas where they often wait for fishing opportunities or " "scraps from nearby boats. The rocky perch and marina setting are typical " "habitats where pelicans congregate along coastlines." ) class Dog(BaseModel): name: str age: int bio: str @pytest.mark.vcr def test_schema_prompt(): model = llm.get_model("claude-sonnet-4.5") response = model.prompt("Invent a good dog", schema=Dog, key=ANTHROPIC_API_KEY) dog = json.loads(response.text()) assert dog == snapshot( { "name": "Biscuit", "age": 4, "bio": ( "Biscuit is a golden retriever with a gentle soul and boundless " "enthusiasm. He greets every person with a wagging tail and has an uncanny " "ability to sense when someone needs comfort. His favorite activities " "include playing fetch at the beach, napping in sunny spots, and stealing " "socks to add to his secret collection under the bed." ), } ) @pytest.mark.vcr @pytest.mark.asyncio async def test_schema_prompt_async(): model = llm.get_async_model("claude-sonnet-4.5") response = await model.prompt( "Invent a terrific dog", schema=Dog, key=ANTHROPIC_API_KEY ) dog_json = await response.text() dog = json.loads(dog_json) assert dog == snapshot( { "name": "Luna", "age": 4, "bio": ( "Luna is a brilliant Golden Retriever with a heart of gold who serves as " "a certified therapy dog at children's hospitals. She has an uncanny " "ability to sense when someone needs comfort and gently rests her head on " "their lap. Luna loves swimming in lakes, playing fetch with her favorite " "tennis ball, and has learned over 50 commands including helping her owner " "retrieve items from around the house." ), } ) @pytest.mark.vcr def test_prompt_with_prefill_and_stop_sequences(): model = llm.get_model("claude-haiku-4.5") response = model.prompt( "Very short function describing a pelican", prefill="```python", stop_sequences=["```"], hide_prefill=True, key=ANTHROPIC_API_KEY, ) text = response.text() assert text == snapshot( "\ndef pelican():\n" ' return "A large waterbird with a long bill and a throat pouch for catching fish."\n' ) @pytest.mark.vcr def test_thinking_prompt(): model = llm.get_model("claude-sonnet-4.5") conversation = model.conversation() response = conversation.prompt( "Two names for a pet pelican, be brief", thinking=True, key=ANTHROPIC_API_KEY ) assert response.text() == snapshot("- Captain\n- Scoop") response_dict = dict(response.response_json) response_dict.pop("id") # differs between requests # Check structure without exact thinking signature assert response_dict["model"] == snapshot("claude-sonnet-4-5-20250929") assert response_dict["stop_reason"] == snapshot("end_turn") content_types = [block["type"] for block in response_dict["content"]] assert "thinking" in content_types assert "text" in content_types assert response.input_tokens == snapshot(46) assert response.output_tokens == snapshot(84) assert response.token_details is None @pytest.mark.vcr def test_tools(): model = llm.get_model("claude-haiku-4.5") names = ["Charles", "Sammy"] chain_response = model.chain( "Two names for a pet pelican", tools=[ llm.Tool.function(lambda: names.pop(0), name="pelican_name_generator"), ], key=ANTHROPIC_API_KEY, ) text = chain_response.text() assert text == snapshot( " Here are two great names for your pet pelican:\n\n" "1. **Charles** - A sophisticated and dignified name, perfect for a pelican with personality!\n" "2. **Sammy** - A friendly and playful name that gives off warm, approachable vibes.\n\n" "Either of these would make an excellent name for your feathered friend! \U0001f985" ) tool_calls = chain_response._responses[0].tool_calls() assert len(tool_calls) == 2 assert all(call.name == "pelican_name_generator" for call in tool_calls) assert [ result.output for result in chain_response._responses[1].prompt.tool_results ] == snapshot(["Charles", "Sammy"]) @pytest.mark.vcr def test_fixed_version_tool_chain_regression(): model = llm.get_model("claude-haiku-4.5") fixed_version = fixed_version_tool() chain_response = model.chain( "Use the fixed_version tool. Then tell me the version and make one short joke about it.", tools=[fixed_version], key=ANTHROPIC_API_KEY, ) text = chain_response.text() assert FIXED_TEST_VERSION in text assert len(chain_response._responses) == 2 second_response = chain_response._responses[1] assert second_response.prompt.tool_results[0].output == FIXED_TEST_VERSION second_request_messages = model.build_messages( second_response.prompt, second_response.conversation ) assert [message["role"] for message in second_request_messages] == [ "user", "assistant", "user", ] assert second_request_messages[1]["content"][-1]["type"] == "tool_use" assert [block["type"] for block in second_request_messages[2]["content"]] == [ "tool_result" ] @pytest.mark.vcr def test_fixed_version_tool_chain_with_thinking_display_regression(): model = llm.get_model("claude-haiku-4.5") from llm.parts import ReasoningPart fixed_version = fixed_version_tool() chain_response = model.chain( "Use the fixed_version tool. Then tell me the version and make one short joke about it. Think about it first.", tools=[fixed_version], key=ANTHROPIC_API_KEY, options={"thinking": True}, ) text = chain_response.text() assert FIXED_TEST_VERSION in text assert len(chain_response._responses) == 2 first_response = chain_response._responses[0] reasoning_parts = [ part for message in first_response.messages() for part in message.parts if isinstance(part, ReasoningPart) ] assert reasoning_parts[0].provider_metadata["anthropic"]["signature"] second_response = chain_response._responses[1] second_request_messages = model.build_messages( second_response.prompt, second_response.conversation ) assert second_request_messages[1]["content"][0]["type"] == "thinking" assert second_request_messages[1]["content"][0]["signature"] assert second_request_messages[1]["content"][-1]["type"] == "tool_use" assert [block["type"] for block in second_request_messages[2]["content"]] == [ "tool_result" ] @pytest.mark.vcr def test_web_search(): model = llm.get_model("claude-opus-4.1") model.key = model.key or ANTHROPIC_API_KEY from llm_anthropic import WebSearch response = model.prompt( "What is the current weather in San Francisco?", tools=[WebSearch()] ) response_text = str(response) assert len(response_text) > 0 assert any( word in response_text.lower() for word in ["weather", "temperature", "san francisco", "degree", "forecast"] ) response_dict = dict(response.response_json) assert "content" in response_dict assert len(response_dict["content"]) > 0 def test_fast_mode_kwargs(): model = llm.get_model("claude-opus-4.8") prompt = llm.Prompt("Hi", model, options=model.Options(fast=True)) kwargs = model.build_kwargs(prompt, None) assert kwargs["speed"] == "fast" assert "fast-mode-2026-02-01" in kwargs["betas"] def test_fast_mode_off_by_default(): model = llm.get_model("claude-opus-4.8") prompt = llm.Prompt("Hi", model, options=model.Options()) kwargs = model.build_kwargs(prompt, None) assert "speed" not in kwargs assert "betas" not in kwargs def test_opus_5_registered(): model = llm.get_model("claude-opus-5") assert model.model_id == "anthropic/claude-opus-5" assert model.claude_model_id == "claude-opus-5" assert "application/pdf" in model.attachment_types assert model.supports_thinking assert model.supports_thinking_effort assert model.supports_adaptive_thinking assert model.supports_web_search assert model.use_structured_outputs assert model.default_max_tokens == 128000 async_model = llm.get_async_model("claude-opus-5") assert async_model.model_id == "anthropic/claude-opus-5" assert async_model.claude_model_id == "claude-opus-5" def test_opus_5_kwargs(): model = llm.get_model("claude-opus-5") prompt = llm.Prompt("Hi", model, options=model.Options(thinking_effort="max")) kwargs = model.build_kwargs(prompt, None) assert kwargs["model"] == "claude-opus-5" assert kwargs["max_tokens"] == 128000 assert kwargs["thinking"] == {"type": "adaptive"} assert kwargs["output_config"]["effort"] == "max" assert "betas" not in kwargs @pytest.mark.vcr def test_opus_46_prompt(): model = llm.get_model("claude-opus-4.6") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt("Two names for a pet pelican, be brief") text = response.text() assert len(text) > 0 response_dict = dict(response.response_json) assert response_dict["model"] == snapshot("claude-opus-4-6") assert response.input_tokens > 0 assert response.output_tokens > 0 @pytest.mark.vcr def test_sonnet_46_prompt(): model = llm.get_model("claude-sonnet-4.6") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt("Two names for a pet pelican, be brief") text = response.text() assert len(text) > 0 response_dict = dict(response.response_json) assert response_dict["model"] == snapshot("claude-sonnet-4-6") assert response.input_tokens > 0 assert response.output_tokens > 0 @pytest.mark.vcr def test_opus_46_adaptive_thinking(): model = llm.get_model("claude-opus-4.6") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt("Two names for a pet pelican, be brief", thinking=True) text = response.text() assert len(text) > 0 response_dict = dict(response.response_json) # Should have thinking content in the response content_types = [block["type"] for block in response_dict["content"]] assert "thinking" in content_types assert "text" in content_types @pytest.mark.vcr def test_sonnet_46_effort_without_thinking(): model = llm.get_model("claude-sonnet-4.6") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt( "Two names for a pet pelican, be brief", thinking_effort="low" ) text = response.text() assert len(text) > 0 def test_46_prefill_rejected(): model = llm.get_model("claude-opus-4.6") model.key = "test-key" with pytest.raises( ValueError, match="Prefilling assistant messages is not supported" ): model.prompt("Hello", prefill="{").text() def test_max_effort_passed_through(): # Client-side validation of effort levels was removed - the API is # left to reject models that don't support a given level model = llm.get_model("claude-sonnet-4.6") prompt = llm.Prompt("Hello", model, options=model.Options(thinking_effort="max")) kwargs = model.build_kwargs(prompt, None) assert kwargs["output_config"]["effort"] == "max" @pytest.mark.vcr def test_opus_46_schema(): model = llm.get_model("claude-opus-4.6") response = model.prompt("Invent a good dog", schema=Dog, key=ANTHROPIC_API_KEY) dog = json.loads(response.text()) assert "name" in dog assert "age" in dog assert "bio" in dog # Phase 3: StreamEvent tests from llm.parts import StreamEvent, TextPart, ReasoningPart, ToolCallPart, ToolResultPart @pytest.mark.vcr def test_stream_events_text(): """stream_events() yields text StreamEvents.""" model = llm.get_model("claude-haiku-4.5") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt("Say just hello") events = list(response.stream_events()) text_events = [e for e in events if e.type == "text"] assert len(text_events) > 0 text = "".join(e.chunk for e in text_events) assert "hello" in text.lower() or "Hello" in text @pytest.mark.vcr def test_stream_events_thinking(): """stream_events() yields reasoning StreamEvents for thinking.""" model = llm.get_model("claude-haiku-4.5") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt("Two names for a pet pelican, be brief", thinking=True) events = list(response.stream_events()) reasoning_events = [e for e in events if e.type == "reasoning"] text_events = [e for e in events if e.type == "text"] assert len(reasoning_events) > 0, "Should have reasoning events" assert len(text_events) > 0, "Should have text events" # Reasoning should be in earlier part_index than text assert reasoning_events[0].part_index < text_events[0].part_index @pytest.mark.vcr def test_parts_thinking(): """response.parts includes ReasoningPart and TextPart for thinking responses.""" model = llm.get_model("claude-haiku-4.5") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt("Two names for a pet pelican, be brief", thinking=True) response.text() parts = [p for m in response.messages() for p in m.parts] reasoning_parts = [p for p in parts if isinstance(p, ReasoningPart)] text_parts = [p for p in parts if isinstance(p, TextPart)] assert len(reasoning_parts) >= 1, "Should have reasoning part" assert len(text_parts) >= 1, "Should have text part" assert reasoning_parts[0].provider_metadata["anthropic"]["signature"] assert reasoning_parts[0].text, "Reasoning text should not be empty" assert text_parts[0].text, "Text should not be empty" @pytest.mark.vcr def test_stream_events_tool_calls(): """stream_events() yields tool call StreamEvents.""" model = llm.get_model("claude-haiku-4.5") model.key = model.key or ANTHROPIC_API_KEY names = ["Charles"] response = model.prompt( "Generate one name for a pet pelican", tools=[ llm.Tool.function(lambda: names.pop(0), name="pelican_name_generator"), ], key=ANTHROPIC_API_KEY, ) events = list(response.stream_events()) name_events = [e for e in events if e.type == "tool_call_name"] args_events = [e for e in events if e.type == "tool_call_args"] assert len(name_events) >= 1, "Should have tool_call_name event" assert name_events[0].chunk == snapshot("pelican_name_generator") assert name_events[0].tool_call_id is not None @pytest.mark.vcr def test_web_search_tool_result_ordering(): """web_search_tool_result parts appear BEFORE the text that uses them.""" model = llm.get_model("claude-opus-4.1") model.key = model.key or ANTHROPIC_API_KEY from llm_anthropic import WebSearch response = model.prompt( "What is the current weather in San Francisco?", tools=[WebSearch()] ) events = list(response.stream_events()) # Find indices of first tool_result and first text event tool_result_indices = [i for i, e in enumerate(events) if e.type == "tool_result"] text_indices = [ i for i, e in enumerate(events) if e.type == "text" and e.chunk.strip() # non-empty text ] assert len(tool_result_indices) >= 1, "Should have tool_result events" assert len(text_indices) >= 1, "Should have text events" # The tool_result should come before the main text content first_tool_result = tool_result_indices[0] first_substantive_text = text_indices[0] assert first_tool_result < first_substantive_text, ( f"tool_result at index {first_tool_result} should come before " f"text at index {first_substantive_text}" ) # Also verify via parts parts = [p for m in response.messages() for p in m.parts] part_types = [type(p).__name__ for p in parts] # ToolResultPart should appear before the main TextParts if "ToolResultPart" in part_types and "TextPart" in part_types: first_result_idx = part_types.index("ToolResultPart") first_text_idx = part_types.index("TextPart") assert first_result_idx < first_text_idx # --- messages= parameter -------------------------------------------------- # # Unit tests that exercise build_messages directly on messages= input. # Pure structural — no API calls, so no cassettes. def _build_messages_for(prompt_kwargs): """Invoke build_messages on a one-shot Prompt without hitting the API.""" model = llm.get_model("claude-sonnet-4.5") options = prompt_kwargs.pop("options", model.Options()) p = llm.Prompt(None, model=model, options=options, **prompt_kwargs) return model.build_messages(p, None) def test_build_messages_simple_user_text(): from llm import user msgs = _build_messages_for({"messages": [user("hi")]}) assert msgs == [{"role": "user", "content": [{"type": "text", "text": "hi"}]}] def test_build_messages_skips_system_role(): from llm import system, user msgs = _build_messages_for({"messages": [system("be nice"), user("hi")]}) # System does not appear in the messages list; it goes to kwargs["system"]. assert msgs == [{"role": "user", "content": [{"type": "text", "text": "hi"}]}] def test_build_messages_merges_tool_then_user(): """A tool-role message followed by a user message must collapse into one Anthropic user turn (tool_result + text in the same content array).""" from llm import user, tool_message from llm.parts import ToolResultPart msgs = _build_messages_for( { "messages": [ tool_message( ToolResultPart(name="search", output="sunny", tool_call_id="call_1") ), user("thanks"), ] } ) assert msgs == [ { "role": "user", "content": [ { "type": "tool_result", "tool_use_id": "call_1", "content": "sunny", }, {"type": "text", "text": "thanks"}, ], } ] def test_build_messages_assistant_tool_call_and_text(): from llm import assistant, user from llm.parts import TextPart, ToolCallPart msgs = _build_messages_for( { "messages": [ user("what time?"), assistant( TextPart(text="Let me check"), ToolCallPart(name="clock", arguments={}, tool_call_id="c1"), ), ] } ) assert msgs == [ {"role": "user", "content": [{"type": "text", "text": "what time?"}]}, { "role": "assistant", "content": [ {"type": "text", "text": "Let me check"}, { "type": "tool_use", "id": "c1", "name": "clock", "input": {}, }, ], }, ] def test_build_messages_reasoning_round_trips_signature(): """Thinking blocks from a prior assistant message must preserve the Anthropic signature via provider_metadata — otherwise continuation requests involving signed thinking get rejected by the API.""" from llm import assistant, user from llm.parts import ReasoningPart, TextPart msgs = _build_messages_for( { "messages": [ user("q"), assistant( ReasoningPart( text="thinking...", provider_metadata={"anthropic": {"signature": "sig-abc"}}, ), TextPart(text="answer"), ), ] } ) assert msgs[1]["content"][0] == { "type": "thinking", "thinking": "thinking...", "signature": "sig-abc", } def test_load_conversation_preserves_logged_tool_chain_for_anthropic(tmp_path): """Regression for llm -c after a logged tool call chain. LLM 0.32a0 rehydrates the final tool-result response as if its prompt.messages started with only the current tool_result turn. That makes Anthropic reject the next continuation because the request starts with an orphan tool_result instead of the preceding assistant tool_use. """ import datetime import sqlite_utils from llm.cli import load_conversation from llm.migrations import migrate from llm.parts import ToolCallPart, ToolResultPart model = llm.get_model("claude-haiku-4.5") def tick() -> str: return "tock" tool = llm.Tool.function(tick, name="tick") conversation = model.conversation() def mark_done(response): response._done = True response._start = 0.0 response._end = 0.0 response._start_utcnow = datetime.datetime.now(datetime.timezone.utc) first = llm.Response( llm.Prompt("q1", model=model, tools=[tool], options=model.Options()), model, stream=False, conversation=conversation, ) first.add_tool_call(llm.ToolCall(name="tick", arguments={}, tool_call_id="c1")) mark_done(first) tool_result = llm.ToolResult(name="tick", output="tock", tool_call_id="c1") second_chain = [ llm.user("q1"), llm.Message( role="assistant", parts=[ToolCallPart(name="tick", arguments={}, tool_call_id="c1")], ), llm.Message( role="tool", parts=[ToolResultPart(name="tick", output="tock", tool_call_id="c1")], ), ] second = llm.Response( llm.Prompt( "", model=model, tools=[tool], tool_results=[tool_result], messages=second_chain, options=model.Options(), ), model, stream=False, conversation=conversation, ) second._chunks = ["final answer"] second._stream_events = [llm.parts.StreamEvent(type="text", chunk="final answer")] mark_done(second) db = sqlite_utils.Database(str(tmp_path / "logs.db")) migrate(db) first.log_to_db(db) conversation.responses.append(first) second.log_to_db(db) conversation.responses.append(second) loaded = load_conversation(None, database=str(tmp_path / "logs.db")) follow_up = loaded.prompt("q2") anthropic_messages = model.build_messages(follow_up.prompt, loaded) assert anthropic_messages[0]["content"] == [{"type": "text", "text": "q1"}] assert anthropic_messages[1]["content"] == [ {"type": "tool_use", "id": "c1", "name": "tick", "input": {}} ] assert anthropic_messages[2]["content"] == [ {"type": "tool_result", "tool_use_id": "c1", "content": "tock"} ] assert anthropic_messages[3]["content"] == [ {"type": "text", "text": "final answer"} ] assert anthropic_messages[4]["content"] == [{"type": "text", "text": "q2"}] def test_extract_system_from_messages(): from llm import system, user model = llm.get_model("claude-sonnet-4.5") p = llm.Prompt(None, model=model, messages=[system("be helpful"), user("hi")]) assert model._extract_system(p) == "be helpful" def test_extract_system_prefers_prompt_system_over_messages(): """When both paths are populated (synthesized case), prompt.system wins since it already composes system= + system_fragments.""" from llm import user model = llm.get_model("claude-sonnet-4.5") p = llm.Prompt(None, model=model, system="legacy sys", messages=[user("hi")]) assert model._extract_system(p) == "legacy sys" # --- WebFetch server-side tool -------------------------------------------- def test_web_fetch_supported_server_side_tools(): from llm_anthropic import WebFetch model = llm.get_model("claude-sonnet-4.6") assert WebFetch in model.supported_server_side_tools # Models without web search support don't claim WebFetch either old_model = llm.get_model("claude-3-opus") assert WebFetch not in old_model.supported_server_side_tools def test_web_fetch_kwargs(): from llm_anthropic import WebFetch model = llm.get_model("claude-sonnet-4.6") prompt = llm.Prompt( "Fetch https://www.example.com/", model, options=model.Options(), tools=[ WebFetch( max_uses=3, allowed_domains=["example.com"], citations=True, max_content_tokens=5000, ) ], ) kwargs = model.build_kwargs(prompt, None) assert kwargs["tools"] == [ { "type": "web_fetch_20260318", "name": "web_fetch", "max_uses": 3, "allowed_domains": ["example.com"], "citations": {"enabled": True}, "max_content_tokens": 5000, } ] def test_web_fetch_kwargs_basic_version_on_older_model(): from llm_anthropic import WebFetch model = llm.get_model("claude-opus-4.1") prompt = llm.Prompt( "Fetch https://www.example.com/", model, options=model.Options(), tools=[WebFetch()], ) kwargs = model.build_kwargs(prompt, None) assert kwargs["tools"] == [{"type": "web_fetch_20250910", "name": "web_fetch"}] def test_web_fetch_use_cache_requires_newer_model(): from llm_anthropic import WebFetch model = llm.get_model("claude-opus-4.1") prompt = llm.Prompt( "Fetch https://www.example.com/", model, options=model.Options(), tools=[WebFetch(use_cache=False)], ) with pytest.raises(ValueError): model.build_kwargs(prompt, None) # Fine on a 4.6 model model46 = llm.get_model("claude-sonnet-4.6") prompt46 = llm.Prompt( "Fetch https://www.example.com/", model46, options=model46.Options(), tools=[WebFetch(use_cache=False)], ) kwargs = model46.build_kwargs(prompt46, None) assert kwargs["tools"][0]["use_cache"] is False def test_web_fetch_unsupported_model_raises(): from llm_anthropic import WebFetch model = llm.get_model("claude-3-opus") prompt = llm.Prompt( "Fetch https://www.example.com/", model, options=model.Options(), tools=[WebFetch()], ) with pytest.raises(ValueError, match="does not support server-side tool"): model.build_kwargs(prompt, None) def test_web_fetch_alongside_client_tools_kwargs(): from llm_anthropic import WebFetch model = llm.get_model("claude-sonnet-4.6") prompt = llm.Prompt( "Fetch and report version", model, options=model.Options(), tools=[WebFetch(max_uses=1), fixed_version_tool()], ) kwargs = model.build_kwargs(prompt, None) types_and_names = [(tool.get("type"), tool.get("name")) for tool in kwargs["tools"]] assert ("web_fetch_20260318", "web_fetch") in types_and_names assert (None, "fixed_version") in types_and_names @pytest.mark.vcr def test_web_fetch(): from llm_anthropic import WebFetch model = llm.get_model("claude-sonnet-4.6") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt( "Fetch https://www.example.com/ and quote the first sentence of the page", tools=[WebFetch(max_uses=1)], ) text = str(response) assert "example" in text.lower() # Server-executed tool call and result should be captured as parts. # Dynamic filtering means the model may also make code_execution # server tool calls around the fetch itself. parts = [p for m in response.messages() for p in m.parts] tool_calls = [p for p in parts if isinstance(p, llm.parts.ToolCallPart)] web_fetch_calls = [p for p in tool_calls if p.name == "web_fetch"] assert web_fetch_calls assert web_fetch_calls[0].server_executed fetch_results = [] for part in parts: if not isinstance(part, llm.parts.ToolResultPart) or not part.output: continue try: data = json.loads(part.output) except ValueError: continue if isinstance(data, dict) and data.get("type") == "web_fetch_result": fetch_results.append(data) assert fetch_results assert fetch_results[0]["url"] == "https://www.example.com/" def test_build_messages_replays_server_tool_blocks(): """Server-executed tool calls/results replay as server_tool_use and provider result blocks inside the assistant turn - not as client tool_use/tool_result blocks, which the API rejects.""" from llm.parts import Message, TextPart, ToolCallPart, ToolResultPart from llm import user model = llm.get_model("claude-sonnet-4.6") result_payload = { "type": "web_fetch_result", "url": "https://www.example.com/", "content": {"type": "document"}, } msgs = [ user("Fetch https://www.example.com/ and tell me its title"), Message( role="assistant", parts=[ ToolCallPart( name="web_fetch", arguments={"url": "https://www.example.com/"}, tool_call_id="srvtoolu_123", server_executed=True, ), ToolResultPart( name="web_fetch", output=json.dumps(result_payload), tool_call_id="srvtoolu_123", server_executed=True, ), TextPart(text="The title is Example Domain"), ], ), user("What domain was that page on?"), ] prompt = llm.Prompt(None, model, options=model.Options(), messages=msgs) messages = model.build_messages(prompt, None) assert [m["role"] for m in messages] == ["user", "assistant", "user"] assistant_blocks = messages[1]["content"] assert [b["type"] for b in assistant_blocks] == [ "server_tool_use", "web_fetch_tool_result", "text", ] assert assistant_blocks[0]["name"] == "web_fetch" assert assistant_blocks[0]["id"] == "srvtoolu_123" assert assistant_blocks[1]["tool_use_id"] == "srvtoolu_123" assert assistant_blocks[1]["content"] == result_payload def test_sonnet_and_haiku_4_5_support_web_search(): from llm_anthropic import WebFetch for model_id in ("claude-sonnet-4.5", "claude-haiku-4.5"): model = llm.get_model(model_id) assert model.supports_web_search assert WebFetch in model.supported_server_side_tools def test_sonnet_5_registered(): model = llm.get_model("claude-sonnet-5") assert model.model_id == "anthropic/claude-sonnet-5" assert model.claude_model_id == "claude-sonnet-5" assert "application/pdf" in model.attachment_types assert model.supports_thinking assert model.supports_thinking_effort assert model.supports_adaptive_thinking assert model.supports_web_search assert model.use_structured_outputs assert model.default_max_tokens == 128000 async_model = llm.get_async_model("claude-sonnet-5") assert async_model.supports_web_search assert async_model.default_max_tokens == 128000 # --- WebSearch server-side tool ------------------------------------------- def test_web_search_tool_supported_server_side_tools(): from llm_anthropic import WebSearch model = llm.get_model("claude-sonnet-4.6") assert WebSearch in model.supported_server_side_tools old_model = llm.get_model("claude-3-opus") assert WebSearch not in old_model.supported_server_side_tools def test_web_search_tool_kwargs(): from llm_anthropic import WebSearch model = llm.get_model("claude-sonnet-4.6") prompt = llm.Prompt( "What is the weather in London?", model, options=model.Options(), tools=[ WebSearch( max_uses=2, allowed_domains=["example.com"], user_location={"city": "London", "country": "GB"}, ) ], ) kwargs = model.build_kwargs(prompt, None) assert kwargs["tools"] == [ { "type": "web_search_20260318", "name": "web_search", "max_uses": 2, "allowed_domains": ["example.com"], "user_location": { "type": "approximate", "city": "London", "country": "GB", }, } ] def test_web_search_tool_kwargs_basic_version_on_older_model(): from llm_anthropic import WebSearch model = llm.get_model("claude-opus-4.1") prompt = llm.Prompt( "Search the web", model, options=model.Options(), tools=[WebSearch()] ) kwargs = model.build_kwargs(prompt, None) assert kwargs["tools"] == [{"type": "web_search_20250305", "name": "web_search"}] def test_web_search_tool_domain_conflict(): from llm_anthropic import WebSearch with pytest.raises(ValueError): WebSearch(allowed_domains=["a.com"], blocked_domains=["b.com"]) @pytest.mark.vcr def test_web_search_server_side_tool(): from llm_anthropic import WebSearch model = llm.get_model("claude-sonnet-4.6") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt( "What is the current weather in San Francisco?", tools=[WebSearch(max_uses=2)], ) text = str(response) assert any( word in text.lower() for word in ["weather", "temperature", "san francisco", "degree", "forecast"] ) parts = [p for m in response.messages() for p in m.parts] search_calls = [ p for p in parts if isinstance(p, llm.parts.ToolCallPart) and p.name == "web_search" ] assert search_calls assert search_calls[0].server_executed # --- CodeExecution server-side tool --------------------------------------- def test_code_execution_supported_server_side_tools(): from llm_anthropic import CodeExecution for model_id in ("claude-sonnet-4.6", "claude-haiku-4.5", "claude-opus-5"): model = llm.get_model(model_id) assert CodeExecution in model.supported_server_side_tools # Web search capable but too old for code execution old_model = llm.get_model("claude-opus-4.1") assert old_model.supports_web_search assert CodeExecution not in old_model.supported_server_side_tools def test_code_execution_kwargs(): from llm_anthropic import CodeExecution model = llm.get_model("claude-sonnet-4.6") prompt = llm.Prompt( "Compute something", model, options=model.Options(), tools=[CodeExecution()], ) kwargs = model.build_kwargs(prompt, None) assert kwargs["tools"] == [ {"type": "code_execution_20260521", "name": "code_execution"} ] assert "container" not in kwargs def test_code_execution_container_reuse_kwargs(): from llm_anthropic import CodeExecution model = llm.get_model("claude-sonnet-4.6") prompt = llm.Prompt( "Compute something", model, options=model.Options(), tools=[CodeExecution(container="cntr_123")], ) kwargs = model.build_kwargs(prompt, None) assert kwargs["container"] == "cntr_123" def test_code_execution_invalid_container(): from llm_anthropic import CodeExecution with pytest.raises(ValueError): CodeExecution(container=123) def test_code_execution_unsupported_model_raises(): from llm_anthropic import CodeExecution model = llm.get_model("claude-opus-4.1") prompt = llm.Prompt( "Compute something", model, options=model.Options(), tools=[CodeExecution()] ) with pytest.raises(ValueError, match="does not support server-side tool"): model.build_kwargs(prompt, None) @pytest.mark.vcr def test_code_execution(): from llm_anthropic import CodeExecution model = llm.get_model("claude-sonnet-4.6") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt( "Use code execution to compute 123456789 * 987654321 exactly. " "Reply with just the number.", tools=[CodeExecution()], ) assert "121932631112635269" in str(response) parts = [p for m in response.messages() for p in m.parts] exec_calls = [ p for p in parts if isinstance(p, llm.parts.ToolCallPart) and "code_execution" in p.name ] assert exec_calls assert exec_calls[0].server_executed exec_results = [ p for p in parts if isinstance(p, llm.parts.ToolResultPart) and "code_execution" in p.name ] assert exec_results # Container ID must survive streaming (the SDK accumulator drops it) # and be JSON-serializable container = response.response_json["container"] assert container["id"].startswith("container_") assert isinstance(container["expires_at"], str) json.dumps(container) # --- MCP connector server-side tool --------------------------------------- def test_mcp_supported_server_side_tools(): from llm_anthropic import AnthropicMCP model = llm.get_model("claude-sonnet-4.6") assert AnthropicMCP in model.supported_server_side_tools old_model = llm.get_model("claude-3-opus") assert AnthropicMCP not in old_model.supported_server_side_tools def test_mcp_kwargs(): from llm_anthropic import AnthropicMCP model = llm.get_model("claude-sonnet-4.6") prompt = llm.Prompt( "What does simonw/llm do?", model, options=model.Options(), tools=[AnthropicMCP(url="https://mcp.deepwiki.com/mcp", name="deepwiki")], ) kwargs = model.build_kwargs(prompt, None) assert kwargs["tools"] == [{"type": "mcp_toolset", "mcp_server_name": "deepwiki"}] assert kwargs["mcp_servers"] == [ {"type": "url", "url": "https://mcp.deepwiki.com/mcp", "name": "deepwiki"} ] assert kwargs["betas"] == ["mcp-client-2025-11-20"] def test_mcp_name_derived_from_url_host(): from llm_anthropic import AnthropicMCP tool = AnthropicMCP(url="https://mcp.deepwiki.com/mcp") assert tool.server_name == "mcp.deepwiki.com" def test_mcp_kwargs_authorization_token_and_allowed_tools(): from llm_anthropic import AnthropicMCP model = llm.get_model("claude-sonnet-4.6") prompt = llm.Prompt( "Ask a question", model, options=model.Options(), tools=[ AnthropicMCP( url="https://mcp.example.com/mcp", name="example", authorization_token="secret-token", allowed_tools=["ask_question", "read_wiki_structure"], ) ], ) kwargs = model.build_kwargs(prompt, None) assert kwargs["tools"] == [ { "type": "mcp_toolset", "mcp_server_name": "example", "default_config": {"enabled": False}, "configs": { "ask_question": {"enabled": True}, "read_wiki_structure": {"enabled": True}, }, } ] assert kwargs["mcp_servers"] == [ { "type": "url", "url": "https://mcp.example.com/mcp", "name": "example", "authorization_token": "secret-token", } ] def test_mcp_multiple_servers_beta_appended_once(): from llm_anthropic import AnthropicMCP model = llm.get_model("claude-sonnet-4.6") prompt = llm.Prompt( "Use both servers", model, options=model.Options(), tools=[ AnthropicMCP(url="https://mcp.one.com/mcp", name="one"), AnthropicMCP(url="https://mcp.two.com/mcp", name="two"), ], ) kwargs = model.build_kwargs(prompt, None) assert kwargs["betas"].count("mcp-client-2025-11-20") == 1 assert [server["name"] for server in kwargs["mcp_servers"]] == ["one", "two"] assert [tool["mcp_server_name"] for tool in kwargs["tools"]] == ["one", "two"] def test_mcp_validation_errors(): from llm_anthropic import AnthropicMCP with pytest.raises(ValueError): AnthropicMCP(url="") with pytest.raises(ValueError): AnthropicMCP(url="ftp://example.com/mcp") with pytest.raises(ValueError): AnthropicMCP(url="https://mcp.example.com/mcp", name="") with pytest.raises(ValueError): AnthropicMCP(url="https://mcp.example.com/mcp", allowed_tools="ask_question") with pytest.raises(ValueError): AnthropicMCP(url="https://mcp.example.com/mcp", allowed_tools=[""]) def test_mcp_unsupported_model_raises(): from llm_anthropic import AnthropicMCP model = llm.get_model("claude-3-opus") prompt = llm.Prompt( "Use the MCP server", model, options=model.Options(), tools=[AnthropicMCP(url="https://mcp.deepwiki.com/mcp", name="deepwiki")], ) with pytest.raises(ValueError, match="does not support server-side tool"): model.build_kwargs(prompt, None) def test_build_messages_replays_mcp_blocks(): """MCP tool calls/results replay as mcp_tool_use (with the required server_name field) and mcp_tool_result blocks inside the assistant turn - the API rejects them in any other shape.""" from llm.parts import Message, TextPart, ToolCallPart, ToolResultPart from llm import user model = llm.get_model("claude-sonnet-4.6") result_content = [{"type": "text", "text": "llm is a CLI tool", "citations": None}] msgs = [ user("Use the deepwiki tools to say what simonw/llm does"), Message( role="assistant", parts=[ ToolCallPart( name="ask_question", arguments={"repoName": "simonw/llm", "question": "What is it?"}, tool_call_id="mcptoolu_123", server_executed=True, provider_metadata={"anthropic": {"mcp_server_name": "deepwiki"}}, ), ToolResultPart( name="mcp", output=json.dumps(result_content), tool_call_id="mcptoolu_123", server_executed=True, ), TextPart(text="It is a CLI tool for LLMs"), ], ), user("What tool did you call?"), ] prompt = llm.Prompt(None, model, options=model.Options(), messages=msgs) messages = model.build_messages(prompt, None) assert [m["role"] for m in messages] == ["user", "assistant", "user"] assistant_blocks = messages[1]["content"] assert [b["type"] for b in assistant_blocks] == [ "mcp_tool_use", "mcp_tool_result", "text", ] assert assistant_blocks[0] == { "type": "mcp_tool_use", "id": "mcptoolu_123", "name": "ask_question", "input": {"repoName": "simonw/llm", "question": "What is it?"}, "server_name": "deepwiki", } assert assistant_blocks[1]["tool_use_id"] == "mcptoolu_123" assert assistant_blocks[1]["content"] == result_content @pytest.mark.vcr def test_mcp_server_side_tool(): from llm_anthropic import AnthropicMCP model = llm.get_model("claude-sonnet-4.6") model.key = model.key or ANTHROPIC_API_KEY response = model.prompt( "Use the deepwiki tools to find out what the simonw/llm repo does. " "Reply in one sentence.", tools=[AnthropicMCP(url="https://mcp.deepwiki.com/mcp", name="deepwiki")], ) text = str(response) assert "llm" in text.lower() parts = [p for m in response.messages() for p in m.parts] mcp_calls = [ p for p in parts if isinstance(p, llm.parts.ToolCallPart) and (p.tool_call_id or "").startswith("mcptoolu") ] assert mcp_calls assert mcp_calls[0].server_executed assert mcp_calls[0].provider_metadata == { "anthropic": {"mcp_server_name": "deepwiki"} } mcp_results = [ p for p in parts if isinstance(p, llm.parts.ToolResultPart) and (p.tool_call_id or "").startswith("mcptoolu") ] assert mcp_results assert mcp_results[0].server_executed # --- Simplified thinking options (breaking change) ------------------------ def test_thinking_removed_options_rejected(): model = llm.get_model("claude-sonnet-4.6") for option in ("thinking_budget", "thinking_adaptive", "thinking_display"): with pytest.raises(Exception): model.Options(**{option: 1}) def test_thinking_true_adaptive_on_46(): model = llm.get_model("claude-sonnet-4.6") prompt = llm.Prompt("Hi", model, options=model.Options(thinking=True)) kwargs = model.build_kwargs(prompt, None) assert kwargs["thinking"] == {"type": "adaptive"} def test_thinking_true_enabled_on_pre_46(): model = llm.get_model("claude-opus-4.1") prompt = llm.Prompt("Hi", model, options=model.Options(thinking=True)) kwargs = model.build_kwargs(prompt, None) assert kwargs["thinking"] == {"type": "enabled", "budget_tokens": 1024} def test_thinking_false_sends_disabled(): model = llm.get_model("claude-sonnet-5") prompt = llm.Prompt("Hi", model, options=model.Options(thinking=False)) kwargs = model.build_kwargs(prompt, None) assert kwargs["thinking"] == {"type": "disabled"} def test_thinking_false_fable_raises(): model = llm.get_model("claude-fable-5") prompt = llm.Prompt("Hi", model, options=model.Options(thinking=False)) with pytest.raises(ValueError, match="cannot be disabled"): model.build_kwargs(prompt, None) def test_thinking_unset_sends_no_param(): # 5-family models think by default server-side; we send nothing model = llm.get_model("claude-sonnet-5") prompt = llm.Prompt("Hi", model, options=model.Options()) kwargs = model.build_kwargs(prompt, None) assert "thinking" not in kwargs def test_hide_reasoning_sets_omitted_display(): # Explicit thinking + hide_reasoning -> omitted display model = llm.get_model("claude-sonnet-4.6") prompt = llm.Prompt( "Hi", model, options=model.Options(thinking=True), hide_reasoning=True ) kwargs = model.build_kwargs(prompt, None) assert kwargs["thinking"] == {"type": "adaptive", "display": "omitted"} # Enabled mode also supports omitted old_model = llm.get_model("claude-opus-4.1") old_prompt = llm.Prompt( "Hi", old_model, options=old_model.Options(thinking=True), hide_reasoning=True ) old_kwargs = old_model.build_kwargs(old_prompt, None) assert old_kwargs["thinking"] == { "type": "enabled", "budget_tokens": 1024, "display": "omitted", } def test_hide_reasoning_on_default_thinking_model(): # 5-family thinks by default: -R must still reach the API as omitted model = llm.get_model("claude-opus-5") prompt = llm.Prompt("Hi", model, options=model.Options(), hide_reasoning=True) kwargs = model.build_kwargs(prompt, None) assert kwargs["thinking"] == {"type": "adaptive", "display": "omitted"} # But thinking=False + hide_reasoning -> just disabled prompt_off = llm.Prompt( "Hi", model, options=model.Options(thinking=False), hide_reasoning=True ) kwargs_off = model.build_kwargs(prompt_off, None) assert kwargs_off["thinking"] == {"type": "disabled"} def test_thinking_effort_still_works(): model = llm.get_model("claude-sonnet-5") prompt = llm.Prompt("Hi", model, options=model.Options(thinking_effort="max")) kwargs = model.build_kwargs(prompt, None) assert kwargs["thinking"] == {"type": "adaptive"} assert kwargs["output_config"]["effort"] == "max"