初始化项目版本
This commit is contained in:
565
.agents/skills/claude-api/python/claude-api/README.md
Normal file
565
.agents/skills/claude-api/python/claude-api/README.md
Normal file
@@ -0,0 +1,565 @@
|
||||
# Claude API — Python
|
||||
|
||||
## Installation
|
||||
|
||||
```bash
|
||||
pip install anthropic
|
||||
```
|
||||
|
||||
## Client Initialization
|
||||
|
||||
```python
|
||||
import anthropic
|
||||
|
||||
# Default — resolves credentials from the environment:
|
||||
# ANTHROPIC_API_KEY, or ANTHROPIC_AUTH_TOKEN, or an `ant auth login` profile.
|
||||
# Prefer this for local dev; don't hardcode a key.
|
||||
client = anthropic.Anthropic()
|
||||
|
||||
# Explicit API key (only when you must inject a specific key)
|
||||
client = anthropic.Anthropic(api_key="your-api-key")
|
||||
|
||||
# Async client
|
||||
async_client = anthropic.AsyncAnthropic()
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Client Configuration
|
||||
|
||||
### Per-request overrides
|
||||
|
||||
Use `with_options()` to override client settings for a single call without mutating the client:
|
||||
|
||||
```python
|
||||
client.with_options(timeout=5.0, max_retries=5).messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=1024,
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
)
|
||||
```
|
||||
|
||||
### Timeouts
|
||||
|
||||
Default request timeout is 10 minutes. Pass a float (seconds) or an `httpx.Timeout` for granular control. On timeout the SDK raises `anthropic.APITimeoutError` (and retries per `max_retries`).
|
||||
|
||||
```python
|
||||
import httpx
|
||||
|
||||
client = anthropic.Anthropic(timeout=20.0)
|
||||
client = anthropic.Anthropic(
|
||||
timeout=httpx.Timeout(60.0, read=5.0, write=10.0, connect=2.0),
|
||||
)
|
||||
```
|
||||
|
||||
### Retries
|
||||
|
||||
The SDK auto-retries connection errors, 408, 409, 429, and ≥500 with exponential backoff (default 2 retries). Set `max_retries` on the client or via `with_options()`; `max_retries=0` disables.
|
||||
|
||||
### Async performance (aiohttp backend)
|
||||
|
||||
For high-concurrency async workloads, install `anthropic[aiohttp]` and pass `DefaultAioHttpClient` instead of the default httpx backend:
|
||||
|
||||
```python
|
||||
from anthropic import AsyncAnthropic, DefaultAioHttpClient
|
||||
|
||||
async with AsyncAnthropic(http_client=DefaultAioHttpClient()) as client:
|
||||
...
|
||||
```
|
||||
|
||||
### Custom HTTP client (proxy, base URL)
|
||||
|
||||
Use `DefaultHttpxClient` / `DefaultAsyncHttpxClient` — not raw `httpx.Client` — so the SDK's default timeouts and connection limits are preserved:
|
||||
|
||||
```python
|
||||
from anthropic import Anthropic, DefaultHttpxClient
|
||||
|
||||
client = Anthropic(
|
||||
base_url="http://my.test.server.example.com:8083", # or ANTHROPIC_BASE_URL env var
|
||||
http_client=DefaultHttpxClient(proxy="http://my.test.proxy.example.com"),
|
||||
)
|
||||
```
|
||||
|
||||
### Logging
|
||||
|
||||
Set `ANTHROPIC_LOG=debug` (or `info`) to enable SDK logging via the standard `logging` module.
|
||||
|
||||
---
|
||||
|
||||
## Basic Message Request
|
||||
|
||||
```python
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[
|
||||
{"role": "user", "content": "What is the capital of France?"}
|
||||
]
|
||||
)
|
||||
# response.content is a list of content block objects (TextBlock, ThinkingBlock,
|
||||
# ToolUseBlock, ...). Check .type before accessing .text.
|
||||
for block in response.content:
|
||||
if block.type == "text":
|
||||
print(block.text)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## System Prompts
|
||||
|
||||
```python
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
system="You are a helpful coding assistant. Always provide examples in Python.",
|
||||
messages=[{"role": "user", "content": "How do I read a JSON file?"}]
|
||||
)
|
||||
```
|
||||
|
||||
### Mid-conversation system messages (model-gated)
|
||||
|
||||
For operator instructions that arrive mid-conversation (mode switches, injected state), append `{"role": "system", ...}` to `messages` instead of editing top-level `system` — this preserves the cached prefix and carries operator authority. Must follow a user message (or an `assistant` message ending in server-tool use), and must be either the last entry in `messages` or be followed by an `assistant` turn; cannot be `messages[0]`. Unsupported models return a 400 (`role 'system' is not supported on this model`). See `shared/prompt-caching.md` for when to use this vs. top-level `system`.
|
||||
|
||||
```python
|
||||
response = client.messages.create(
|
||||
model=MODEL_ID, # must support mid-conversation system messages
|
||||
max_tokens=16000,
|
||||
system=[{"type": "text", "text": STABLE_SYSTEM, "cache_control": {"type": "ephemeral"}}],
|
||||
messages=history + [
|
||||
{"role": "user", "content": user_message},
|
||||
{"role": "system", "content": "Terse mode enabled — keep responses under 40 words."},
|
||||
],
|
||||
) # No beta header needed — use regular client.messages.create
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Vision (Images)
|
||||
|
||||
### Base64
|
||||
|
||||
```python
|
||||
import base64
|
||||
|
||||
with open("image.png", "rb") as f:
|
||||
image_data = base64.standard_b64encode(f.read()).decode("utf-8")
|
||||
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "image",
|
||||
"source": {
|
||||
"type": "base64",
|
||||
"media_type": "image/png",
|
||||
"data": image_data
|
||||
}
|
||||
},
|
||||
{"type": "text", "text": "What's in this image?"}
|
||||
]
|
||||
}]
|
||||
)
|
||||
```
|
||||
|
||||
### URL
|
||||
|
||||
```python
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{
|
||||
"type": "image",
|
||||
"source": {
|
||||
"type": "url",
|
||||
"url": "https://example.com/image.png"
|
||||
}
|
||||
},
|
||||
{"type": "text", "text": "Describe this image"}
|
||||
]
|
||||
}]
|
||||
)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Prompt Caching
|
||||
|
||||
Cache large context to reduce costs (up to 90% savings). **Caching is a prefix match** — any byte change anywhere in the prefix invalidates everything after it. For placement patterns, architectural guidance (frozen system prompt, deterministic tool order, where to put volatile content), and the silent-invalidator audit checklist, read `shared/prompt-caching.md`.
|
||||
|
||||
### Automatic Caching (Recommended)
|
||||
|
||||
Use top-level `cache_control` to automatically cache the last cacheable block in the request — no need to annotate individual content blocks:
|
||||
|
||||
```python
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
cache_control={"type": "ephemeral"}, # auto-caches the last cacheable block
|
||||
system="You are an expert on this large document...",
|
||||
messages=[{"role": "user", "content": "Summarize the key points"}]
|
||||
)
|
||||
```
|
||||
|
||||
### Manual Cache Control
|
||||
|
||||
For fine-grained control, add `cache_control` to specific content blocks:
|
||||
|
||||
```python
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
system=[{
|
||||
"type": "text",
|
||||
"text": "You are an expert on this large document...",
|
||||
"cache_control": {"type": "ephemeral"} # default TTL is 5 minutes
|
||||
}],
|
||||
messages=[{"role": "user", "content": "Summarize the key points"}]
|
||||
)
|
||||
|
||||
# With explicit TTL (time-to-live)
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
system=[{
|
||||
"type": "text",
|
||||
"text": "You are an expert on this large document...",
|
||||
"cache_control": {"type": "ephemeral", "ttl": "1h"} # 1 hour TTL
|
||||
}],
|
||||
messages=[{"role": "user", "content": "Summarize the key points"}]
|
||||
)
|
||||
```
|
||||
|
||||
### Verifying Cache Hits
|
||||
|
||||
```python
|
||||
print(response.usage.cache_creation_input_tokens) # tokens written to cache (~1.25x cost)
|
||||
print(response.usage.cache_read_input_tokens) # tokens served from cache (~0.1x cost)
|
||||
print(response.usage.input_tokens) # uncached tokens (full cost)
|
||||
```
|
||||
|
||||
If `cache_read_input_tokens` is zero across repeated identical-prefix requests, a silent invalidator is at work — `datetime.now()` or a UUID in the system prompt, unsorted `json.dumps()`, or a varying tool set. See `shared/prompt-caching.md` for the full audit table.
|
||||
|
||||
---
|
||||
|
||||
## Extended Thinking
|
||||
|
||||
> **Fable 5, Claude Opus 5, Opus 4.8, Opus 4.7, Opus 4.6, and Sonnet 4.6:** Use adaptive thinking. `budget_tokens` is removed on Fable 5, Claude Opus 5, Opus 4.8, and 4.7 (400 if sent); deprecated on Opus 4.6 and Sonnet 4.6.
|
||||
> **Claude Opus 5:** thinking is on by default — omitting `thinking` runs adaptive (`{"type": "adaptive"}` is equivalent), unlike Opus 4.8/4.7 where omitting it meant no thinking. `{"type": "disabled"}` is accepted only at effort `high` or lower; pairing it with `xhigh`/`max` returns a 400.
|
||||
> **Older models:** Use `thinking: {type: "enabled", budget_tokens: N}` (must be < `max_tokens`, min 1024).
|
||||
|
||||
```python
|
||||
# Fable 5 / Claude Opus 5 / Opus 4.8 / 4.7 / 4.6: adaptive thinking (recommended)
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
thinking={"type": "adaptive", "display": "summarized"}, # display opt-in: default is omitted (empty thinking text) on Fable 5 / Mythos 5 / Claude Opus 5 / Opus 4.8 / 4.7
|
||||
output_config={"effort": "high"}, # low | medium | high | xhigh | max
|
||||
messages=[{"role": "user", "content": "Solve this step by step..."}]
|
||||
)
|
||||
|
||||
# Access thinking and response
|
||||
for block in response.content:
|
||||
if block.type == "thinking":
|
||||
print(f"Thinking: {block.thinking}")
|
||||
elif block.type == "text":
|
||||
print(f"Response: {block.text}")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Error Handling
|
||||
|
||||
```python
|
||||
import anthropic
|
||||
|
||||
try:
|
||||
response = client.messages.create(...)
|
||||
except anthropic.BadRequestError as e:
|
||||
print(f"Bad request: {e.message}")
|
||||
except anthropic.AuthenticationError:
|
||||
print("Invalid API key")
|
||||
except anthropic.PermissionDeniedError:
|
||||
print("API key lacks required permissions")
|
||||
except anthropic.NotFoundError:
|
||||
print("Invalid model or endpoint")
|
||||
except anthropic.RateLimitError as e:
|
||||
retry_after = int(e.response.headers.get("retry-after", "60"))
|
||||
print(f"Rate limited. Retry after {retry_after}s.")
|
||||
except anthropic.APIStatusError as e:
|
||||
if e.status_code >= 500:
|
||||
print(f"Server error ({e.status_code}). Retry later.")
|
||||
else:
|
||||
print(f"API error: {e.message}")
|
||||
except anthropic.APIConnectionError:
|
||||
print("Network error. Check internet connection.")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Response Helpers
|
||||
|
||||
Every response object exposes `_request_id` (populated from the `request-id` header) — log it when reporting failures to Anthropic. Despite the underscore prefix, this property is public.
|
||||
|
||||
```python
|
||||
message = client.messages.create(...)
|
||||
print(message._request_id) # req_018EeWyXxfu5pfWkrYcMdjWG
|
||||
print(message.to_json()) # serialize the Pydantic model
|
||||
print(message.to_dict()) # plain dict
|
||||
```
|
||||
|
||||
To access raw headers or other response metadata, use `.with_raw_response`:
|
||||
|
||||
```python
|
||||
raw = client.messages.with_raw_response.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=1024,
|
||||
messages=[{"role": "user", "content": "Hello"}],
|
||||
)
|
||||
print(raw.headers.get("request-id"))
|
||||
message = raw.parse() # the Message object messages.create() would have returned
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Multi-Turn Conversations
|
||||
|
||||
The API is stateless — send the full conversation history each time.
|
||||
|
||||
```python
|
||||
class ConversationManager:
|
||||
"""Manage multi-turn conversations with the Claude API."""
|
||||
|
||||
def __init__(self, client: anthropic.Anthropic, model: str, system: str = None):
|
||||
self.client = client
|
||||
self.model = model
|
||||
self.system = system
|
||||
self.messages = []
|
||||
|
||||
def send(self, user_message: str, **kwargs) -> str:
|
||||
"""Send a message and get a response."""
|
||||
self.messages.append({"role": "user", "content": user_message})
|
||||
|
||||
response = self.client.messages.create(
|
||||
model=self.model,
|
||||
max_tokens=kwargs.get("max_tokens", 16000),
|
||||
system=self.system,
|
||||
messages=self.messages,
|
||||
**kwargs
|
||||
)
|
||||
|
||||
assistant_message = next(
|
||||
(b.text for b in response.content if b.type == "text"), ""
|
||||
)
|
||||
self.messages.append({"role": "assistant", "content": assistant_message})
|
||||
|
||||
return assistant_message
|
||||
|
||||
# Usage
|
||||
conversation = ConversationManager(
|
||||
client=anthropic.Anthropic(),
|
||||
model="claude-opus-5",
|
||||
system="You are a helpful assistant."
|
||||
)
|
||||
|
||||
response1 = conversation.send("My name is Alice.")
|
||||
response2 = conversation.send("What's my name?") # Claude remembers "Alice"
|
||||
```
|
||||
|
||||
**Rules:**
|
||||
|
||||
- Consecutive same-role messages are allowed — the API combines them into a single turn
|
||||
- First message must be `user`
|
||||
- `role: "system"` messages are allowed mid-conversation on supporting models (no beta header needed) — see § Mid-conversation system messages above
|
||||
|
||||
---
|
||||
|
||||
### Compaction (long conversations)
|
||||
|
||||
> **Beta, Fable 5, Claude Opus 5, Opus 4.8, Opus 4.7, Opus 4.6, and Sonnet 4.6.** When conversations approach the 200K context window, compaction automatically summarizes earlier context server-side. The API returns a `compaction` block; you must pass it back on subsequent requests — append `response.content`, not just the text.
|
||||
|
||||
```python
|
||||
import anthropic
|
||||
|
||||
client = anthropic.Anthropic()
|
||||
messages = []
|
||||
|
||||
def chat(user_message: str) -> str:
|
||||
messages.append({"role": "user", "content": user_message})
|
||||
|
||||
response = client.beta.messages.create(
|
||||
betas=["compact-2026-01-12"],
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=messages,
|
||||
context_management={
|
||||
"edits": [{"type": "compact_20260112"}]
|
||||
}
|
||||
)
|
||||
|
||||
# Append full content — compaction blocks must be preserved
|
||||
messages.append({"role": "assistant", "content": response.content})
|
||||
|
||||
return next(block.text for block in response.content if block.type == "text")
|
||||
|
||||
# Compaction triggers automatically when context grows large
|
||||
print(chat("Help me build a Python web scraper"))
|
||||
print(chat("Add support for JavaScript-rendered pages"))
|
||||
print(chat("Now add rate limiting and error handling"))
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Stop Reasons
|
||||
|
||||
The `stop_reason` field in the response indicates why the model stopped generating:
|
||||
|
||||
| Value | Meaning |
|
||||
|-------|---------|
|
||||
| `end_turn` | Claude finished its response naturally |
|
||||
| `max_tokens` | Hit the `max_tokens` limit — increase it or use streaming |
|
||||
| `stop_sequence` | Hit a custom stop sequence |
|
||||
| `tool_use` | Claude wants to call a tool — execute it and continue |
|
||||
| `pause_turn` | Model paused and can be resumed (agentic flows) |
|
||||
| `refusal` | Claude refused for safety reasons — check `stop_details` |
|
||||
|
||||
### Structured Stop Details
|
||||
|
||||
When `stop_reason` is `"refusal"`, the response includes a `stop_details` object with structured information about the refusal:
|
||||
|
||||
```python
|
||||
if response.stop_reason == "refusal" and response.stop_details:
|
||||
print(f"Category: {response.stop_details.category}") # e.g. "cyber", "bio", "reasoning_extraction", "frontier_llm", or None — see docs for the full set
|
||||
print(f"Explanation: {response.stop_details.explanation}")
|
||||
```
|
||||
|
||||
### Refusal Fallbacks (Claude Fable 5) — opt in by default
|
||||
|
||||
Fallbacks are **opt-in**: without them a refused request simply stops. Include the server-side `fallbacks` parameter in `claude-fable-5` code by default — on a policy decline the API re-runs the same request on the fallback model inside the same call. A decline before any output isn't billed (a mid-stream decline bills the streamed partial); the rescue bills at the fallback model's own rates, with cache repricing applied automatically.
|
||||
|
||||
```python
|
||||
response = client.beta.messages.create(
|
||||
model="claude-fable-5",
|
||||
max_tokens=16000,
|
||||
betas=["server-side-fallback-2026-06-01"],
|
||||
fallbacks=[{"model": "claude-opus-4-8"}],
|
||||
messages=[{"role": "user", "content": "..."}],
|
||||
)
|
||||
|
||||
# Switch points: one fallback block per model that ran and declined this turn
|
||||
for block in response.content:
|
||||
if block.type == "fallback":
|
||||
print(f"{block.from_.model} declined; {block.to.model} continued")
|
||||
|
||||
# Served-by signal — covers sticky turns, which carry no fallback block.
|
||||
# Pair with stop_reason: the fallback model can itself refuse.
|
||||
fallback_ran = any(
|
||||
entry.type == "fallback_message" for entry in response.usage.iterations or []
|
||||
)
|
||||
if fallback_ran and response.stop_reason != "refusal":
|
||||
print(f"Served by {response.model}")
|
||||
```
|
||||
|
||||
A `stop_reason: "refusal"` on the final response means the whole chain refused. The header must be exactly `server-side-fallback-2026-06-01` **for this array form**; the newer `fallbacks: "default"` scalar form uses `server-side-fallback-2026-07-01` instead (see `shared/model-migration.md` → Migrating to Claude Opus 5 → New API features), and pairing either header with the other form returns a 400. The parameter is rejected on the Batches API and unavailable on Amazon Bedrock, Vertex AI, and Microsoft Foundry — register the client-side `BetaRefusalFallbackMiddleware` on the client there instead. Full semantics (sticky routing, billing, streaming, echoing fallback turns back): `shared/model-migration.md` → Migrating to Claude Fable 5 → `refusal` stop reason.
|
||||
|
||||
---
|
||||
|
||||
## Cost Optimization Strategies
|
||||
|
||||
### 1. Use Prompt Caching for Repeated Context
|
||||
|
||||
```python
|
||||
# Automatic caching (simplest — caches the last cacheable block)
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
cache_control={"type": "ephemeral"},
|
||||
system=large_document_text, # e.g., 50KB of context
|
||||
messages=[{"role": "user", "content": "Summarize the key points"}]
|
||||
)
|
||||
|
||||
# First request: full cost
|
||||
# Subsequent requests: ~90% cheaper for cached portion
|
||||
```
|
||||
|
||||
### 2. Choose the Right Model
|
||||
|
||||
```python
|
||||
# Default to Opus for most tasks
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5", # $5.00/$25.00 per 1M tokens
|
||||
max_tokens=16000,
|
||||
messages=[{"role": "user", "content": "Explain quantum computing"}]
|
||||
)
|
||||
|
||||
# Use Sonnet for high-volume production workloads
|
||||
standard_response = client.messages.create(
|
||||
model="claude-sonnet-5", # $3.00/$15.00 per 1M tokens
|
||||
max_tokens=16000,
|
||||
messages=[{"role": "user", "content": "Summarize this document"}]
|
||||
)
|
||||
|
||||
# Use Haiku only for simple, speed-critical tasks
|
||||
simple_response = client.messages.create(
|
||||
model="claude-haiku-4-5", # $1.00/$5.00 per 1M tokens
|
||||
max_tokens=256,
|
||||
messages=[{"role": "user", "content": "Classify this as positive or negative"}]
|
||||
)
|
||||
```
|
||||
|
||||
### 3. Use Token Counting Before Requests
|
||||
|
||||
```python
|
||||
count_response = client.messages.count_tokens(
|
||||
model="claude-opus-5",
|
||||
messages=messages,
|
||||
system=system
|
||||
)
|
||||
|
||||
estimated_input_cost = count_response.input_tokens * 0.000005 # $5/1M tokens
|
||||
print(f"Estimated input cost: ${estimated_input_cost:.4f}")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Retry with Exponential Backoff
|
||||
|
||||
> **Note:** The Anthropic SDK automatically retries rate limit (429) and server errors (5xx) with exponential backoff. You can configure this with `max_retries` (default: 2). Only implement custom retry logic if you need behavior beyond what the SDK provides.
|
||||
|
||||
```python
|
||||
import time
|
||||
import random
|
||||
import anthropic
|
||||
|
||||
def call_with_retry(
|
||||
client: anthropic.Anthropic,
|
||||
max_retries: int = 5,
|
||||
base_delay: float = 1.0,
|
||||
max_delay: float = 60.0,
|
||||
**kwargs
|
||||
):
|
||||
"""Call the API with exponential backoff retry."""
|
||||
last_exception = None
|
||||
|
||||
for attempt in range(max_retries):
|
||||
try:
|
||||
return client.messages.create(**kwargs)
|
||||
except anthropic.RateLimitError as e:
|
||||
last_exception = e
|
||||
except anthropic.APIStatusError as e:
|
||||
if e.status_code >= 500:
|
||||
last_exception = e
|
||||
else:
|
||||
raise # Client errors (4xx except 429) should not be retried
|
||||
|
||||
delay = min(base_delay * (2 ** attempt) + random.uniform(0, 1), max_delay)
|
||||
print(f"Retry {attempt + 1}/{max_retries} after {delay:.1f}s")
|
||||
time.sleep(delay)
|
||||
|
||||
raise last_exception
|
||||
```
|
||||
198
.agents/skills/claude-api/python/claude-api/batches.md
Normal file
198
.agents/skills/claude-api/python/claude-api/batches.md
Normal file
@@ -0,0 +1,198 @@
|
||||
# Message Batches API — Python
|
||||
|
||||
The Batches API (`POST /v1/messages/batches`) processes Messages API requests asynchronously at 50% of standard prices.
|
||||
|
||||
## Key Facts
|
||||
|
||||
- Up to 100,000 requests or 256 MB per batch
|
||||
- Most batches complete within 1 hour; maximum 24 hours
|
||||
- Results available for 29 days after creation
|
||||
- 50% cost reduction on all token usage
|
||||
- All Messages API features supported (vision, tools, caching, etc.)
|
||||
|
||||
---
|
||||
|
||||
## Create a Batch
|
||||
|
||||
```python
|
||||
import anthropic
|
||||
from anthropic.types.message_create_params import MessageCreateParamsNonStreaming
|
||||
from anthropic.types.messages.batch_create_params import Request
|
||||
|
||||
client = anthropic.Anthropic()
|
||||
|
||||
message_batch = client.messages.batches.create(
|
||||
requests=[
|
||||
Request(
|
||||
custom_id="request-1",
|
||||
params=MessageCreateParamsNonStreaming(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{"role": "user", "content": "Summarize climate change impacts"}]
|
||||
)
|
||||
),
|
||||
Request(
|
||||
custom_id="request-2",
|
||||
params=MessageCreateParamsNonStreaming(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{"role": "user", "content": "Explain quantum computing basics"}]
|
||||
)
|
||||
),
|
||||
]
|
||||
)
|
||||
|
||||
print(f"Batch ID: {message_batch.id}")
|
||||
print(f"Status: {message_batch.processing_status}")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Poll for Completion
|
||||
|
||||
```python
|
||||
import time
|
||||
|
||||
while True:
|
||||
batch = client.messages.batches.retrieve(message_batch.id)
|
||||
if batch.processing_status == "ended":
|
||||
break
|
||||
print(f"Status: {batch.processing_status}, processing: {batch.request_counts.processing}")
|
||||
time.sleep(60)
|
||||
|
||||
print("Batch complete!")
|
||||
print(f"Succeeded: {batch.request_counts.succeeded}")
|
||||
print(f"Errored: {batch.request_counts.errored}")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Retrieve Results
|
||||
|
||||
> **Note:** Examples below use `match/case` syntax, requiring Python 3.10+. For earlier versions, use `if/elif` chains instead.
|
||||
|
||||
```python
|
||||
for result in client.messages.batches.results(message_batch.id):
|
||||
match result.result.type:
|
||||
case "succeeded":
|
||||
msg = result.result.message
|
||||
text = next((b.text for b in msg.content if b.type == "text"), "")
|
||||
print(f"[{result.custom_id}] {text[:100]}")
|
||||
case "errored":
|
||||
if result.result.error.type == "invalid_request":
|
||||
print(f"[{result.custom_id}] Validation error - fix request and retry")
|
||||
else:
|
||||
print(f"[{result.custom_id}] Server error - safe to retry")
|
||||
case "canceled":
|
||||
print(f"[{result.custom_id}] Canceled")
|
||||
case "expired":
|
||||
print(f"[{result.custom_id}] Expired - resubmit")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Cancel a Batch
|
||||
|
||||
```python
|
||||
cancelled = client.messages.batches.cancel(message_batch.id)
|
||||
print(f"Status: {cancelled.processing_status}") # "canceling"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## List Batches (auto-pagination)
|
||||
|
||||
Iterating the return value of any `list()` call auto-paginates across all pages — do not index into `.data` if you want the full set:
|
||||
|
||||
```python
|
||||
for batch in client.messages.batches.list(limit=20):
|
||||
print(batch.id, batch.processing_status)
|
||||
```
|
||||
|
||||
For manual control, use `first_page.has_next_page()` / `first_page.get_next_page()` / `first_page.next_page_info()`; `first_page.data` holds the current page's items and `first_page.last_id` is the cursor.
|
||||
|
||||
---
|
||||
|
||||
## Batch with Prompt Caching
|
||||
|
||||
```python
|
||||
shared_system = [
|
||||
{"type": "text", "text": "You are a literary analyst."},
|
||||
{
|
||||
"type": "text",
|
||||
"text": large_document_text, # Shared across all requests
|
||||
"cache_control": {"type": "ephemeral"}
|
||||
}
|
||||
]
|
||||
|
||||
message_batch = client.messages.batches.create(
|
||||
requests=[
|
||||
Request(
|
||||
custom_id=f"analysis-{i}",
|
||||
params=MessageCreateParamsNonStreaming(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
system=shared_system,
|
||||
messages=[{"role": "user", "content": question}]
|
||||
)
|
||||
)
|
||||
for i, question in enumerate(questions)
|
||||
]
|
||||
)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Full End-to-End Example
|
||||
|
||||
```python
|
||||
import anthropic
|
||||
import time
|
||||
from anthropic.types.message_create_params import MessageCreateParamsNonStreaming
|
||||
from anthropic.types.messages.batch_create_params import Request
|
||||
|
||||
client = anthropic.Anthropic()
|
||||
|
||||
# 1. Prepare requests
|
||||
items_to_classify = [
|
||||
"The product quality is excellent!",
|
||||
"Terrible customer service, never again.",
|
||||
"It's okay, nothing special.",
|
||||
]
|
||||
|
||||
requests = [
|
||||
Request(
|
||||
custom_id=f"classify-{i}",
|
||||
params=MessageCreateParamsNonStreaming(
|
||||
model="claude-haiku-4-5",
|
||||
max_tokens=50,
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": f"Classify as positive/negative/neutral (one word): {text}"
|
||||
}]
|
||||
)
|
||||
)
|
||||
for i, text in enumerate(items_to_classify)
|
||||
]
|
||||
|
||||
# 2. Create batch
|
||||
batch = client.messages.batches.create(requests=requests)
|
||||
print(f"Created batch: {batch.id}")
|
||||
|
||||
# 3. Wait for completion
|
||||
while True:
|
||||
batch = client.messages.batches.retrieve(batch.id)
|
||||
if batch.processing_status == "ended":
|
||||
break
|
||||
time.sleep(10)
|
||||
|
||||
# 4. Collect results
|
||||
results = {}
|
||||
for result in client.messages.batches.results(batch.id):
|
||||
if result.result.type == "succeeded":
|
||||
msg = result.result.message
|
||||
results[result.custom_id] = next((b.text for b in msg.content if b.type == "text"), "")
|
||||
|
||||
for custom_id, classification in sorted(results.items()):
|
||||
print(f"{custom_id}: {classification}")
|
||||
```
|
||||
170
.agents/skills/claude-api/python/claude-api/files-api.md
Normal file
170
.agents/skills/claude-api/python/claude-api/files-api.md
Normal file
@@ -0,0 +1,170 @@
|
||||
# Files API — Python
|
||||
|
||||
The Files API uploads files for use in Messages API requests. Reference files via `file_id` in content blocks, avoiding re-uploads across multiple API calls.
|
||||
|
||||
**Beta:** Pass `betas=["files-api-2025-04-14"]` in your API calls (the SDK sets the required header automatically).
|
||||
|
||||
## Key Facts
|
||||
|
||||
- Maximum file size: 500 MB
|
||||
- Total storage: 100 GB per organization
|
||||
- Files persist until deleted
|
||||
- File operations (upload, list, delete) are free; content used in messages is billed as input tokens
|
||||
- Not available on Amazon Bedrock or Google Vertex AI
|
||||
|
||||
---
|
||||
|
||||
## Upload a File
|
||||
|
||||
The `file` argument accepts a `(filename, content, content_type)` tuple, a `pathlib.Path` (or any `PathLike` — read for you, async-safe with `AsyncAnthropic`), or an open binary file object.
|
||||
|
||||
```python
|
||||
import anthropic
|
||||
from pathlib import Path
|
||||
|
||||
client = anthropic.Anthropic()
|
||||
|
||||
uploaded = client.beta.files.upload(
|
||||
file=("report.pdf", open("report.pdf", "rb"), "application/pdf"),
|
||||
)
|
||||
# or: client.beta.files.upload(file=Path("report.pdf"))
|
||||
print(f"File ID: {uploaded.id}")
|
||||
print(f"Size: {uploaded.size_bytes} bytes")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Use a File in Messages
|
||||
|
||||
### PDF / Text Document
|
||||
|
||||
```python
|
||||
response = client.beta.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "Summarize the key findings in this report."},
|
||||
{
|
||||
"type": "document",
|
||||
"source": {"type": "file", "file_id": uploaded.id},
|
||||
"title": "Q4 Report", # optional
|
||||
"citations": {"enabled": True} # optional, enables citations
|
||||
}
|
||||
]
|
||||
}],
|
||||
betas=["files-api-2025-04-14"],
|
||||
)
|
||||
for block in response.content:
|
||||
if block.type == "text":
|
||||
print(block.text)
|
||||
```
|
||||
|
||||
### Image
|
||||
|
||||
```python
|
||||
image_file = client.beta.files.upload(
|
||||
file=("photo.png", open("photo.png", "rb"), "image/png"),
|
||||
)
|
||||
|
||||
response = client.beta.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "What's in this image?"},
|
||||
{
|
||||
"type": "image",
|
||||
"source": {"type": "file", "file_id": image_file.id}
|
||||
}
|
||||
]
|
||||
}],
|
||||
betas=["files-api-2025-04-14"],
|
||||
)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Manage Files
|
||||
|
||||
### List Files
|
||||
|
||||
Iterate the list result directly — the SDK auto-paginates across all pages. Only use `.data` if you want the first page only.
|
||||
|
||||
```python
|
||||
for f in client.beta.files.list():
|
||||
print(f"{f.id}: {f.filename} ({f.size_bytes} bytes)")
|
||||
```
|
||||
|
||||
### Get File Metadata
|
||||
|
||||
```python
|
||||
file_info = client.beta.files.retrieve_metadata("file_011CNha8iCJcU1wXNR6q4V8w")
|
||||
print(f"Filename: {file_info.filename}")
|
||||
print(f"MIME type: {file_info.mime_type}")
|
||||
```
|
||||
|
||||
### Delete a File
|
||||
|
||||
```python
|
||||
client.beta.files.delete("file_011CNha8iCJcU1wXNR6q4V8w")
|
||||
```
|
||||
|
||||
### Download a File
|
||||
|
||||
Only files created by the code execution tool or skills can be downloaded (not user-uploaded files).
|
||||
|
||||
```python
|
||||
file_content = client.beta.files.download("file_011CNha8iCJcU1wXNR6q4V8w")
|
||||
file_content.write_to_file("output.txt")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Full End-to-End Example
|
||||
|
||||
Upload a document once, ask multiple questions about it:
|
||||
|
||||
```python
|
||||
import anthropic
|
||||
|
||||
client = anthropic.Anthropic()
|
||||
|
||||
# 1. Upload once
|
||||
uploaded = client.beta.files.upload(
|
||||
file=("contract.pdf", open("contract.pdf", "rb"), "application/pdf"),
|
||||
)
|
||||
print(f"Uploaded: {uploaded.id}")
|
||||
|
||||
# 2. Ask multiple questions using the same file_id
|
||||
questions = [
|
||||
"What are the key terms and conditions?",
|
||||
"What is the termination clause?",
|
||||
"Summarize the payment schedule.",
|
||||
]
|
||||
|
||||
for question in questions:
|
||||
response = client.beta.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": question},
|
||||
{
|
||||
"type": "document",
|
||||
"source": {"type": "file", "file_id": uploaded.id}
|
||||
}
|
||||
]
|
||||
}],
|
||||
betas=["files-api-2025-04-14"],
|
||||
)
|
||||
print(f"\nQ: {question}")
|
||||
text = next((b.text for b in response.content if b.type == "text"), "")
|
||||
print(f"A: {text[:200]}")
|
||||
|
||||
# 3. Clean up when done
|
||||
client.beta.files.delete(uploaded.id)
|
||||
```
|
||||
179
.agents/skills/claude-api/python/claude-api/streaming.md
Normal file
179
.agents/skills/claude-api/python/claude-api/streaming.md
Normal file
@@ -0,0 +1,179 @@
|
||||
# Streaming — Python
|
||||
|
||||
## Quick Start
|
||||
|
||||
```python
|
||||
with client.messages.stream(
|
||||
model="claude-opus-5",
|
||||
max_tokens=64000,
|
||||
messages=[{"role": "user", "content": "Write a story"}]
|
||||
) as stream:
|
||||
for text in stream.text_stream:
|
||||
print(text, end="", flush=True)
|
||||
```
|
||||
|
||||
### Async
|
||||
|
||||
```python
|
||||
async with async_client.messages.stream(
|
||||
model="claude-opus-5",
|
||||
max_tokens=64000,
|
||||
messages=[{"role": "user", "content": "Write a story"}]
|
||||
) as stream:
|
||||
async for text in stream.text_stream:
|
||||
print(text, end="", flush=True)
|
||||
```
|
||||
|
||||
### Low-level: `stream=True`
|
||||
|
||||
`messages.stream()` (above) is the recommended helper — it accumulates state and exposes `text_stream` / `get_final_message()`. If you only need the raw event iterator and want lower memory use, pass `stream=True` to `messages.create()` instead:
|
||||
|
||||
```python
|
||||
for event in client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=64000,
|
||||
messages=[{"role": "user", "content": "Write a story"}],
|
||||
stream=True,
|
||||
):
|
||||
print(event.type)
|
||||
```
|
||||
|
||||
No final-message accumulation is done for you in this form.
|
||||
|
||||
---
|
||||
|
||||
## Handling Different Content Types
|
||||
|
||||
Claude may return text, thinking blocks, or tool use. Handle each appropriately:
|
||||
|
||||
> **Fable 5 / Claude Opus 5 / Opus 4.8 / Opus 4.7 / Opus 4.6:** Use `thinking: {type: "adaptive"}`. On Claude Opus 5 adaptive is also what you get by omitting `thinking` entirely. On older models, use `thinking: {type: "enabled", budget_tokens: N}` instead.
|
||||
|
||||
```python
|
||||
with client.messages.stream(
|
||||
model="claude-opus-5",
|
||||
max_tokens=64000,
|
||||
thinking={"type": "adaptive", "display": "summarized"}, # display opt-in: default is omitted (empty thinking text) on Fable 5 / Mythos 5 / Claude Opus 5 / Opus 4.8 / 4.7
|
||||
messages=[{"role": "user", "content": "Analyze this problem"}]
|
||||
) as stream:
|
||||
for event in stream:
|
||||
if event.type == "content_block_start":
|
||||
if event.content_block.type == "thinking":
|
||||
print("\n[Thinking...]")
|
||||
elif event.content_block.type == "text":
|
||||
print("\n[Response:]")
|
||||
|
||||
elif event.type == "content_block_delta":
|
||||
if event.delta.type == "thinking_delta":
|
||||
print(event.delta.thinking, end="", flush=True)
|
||||
elif event.delta.type == "text_delta":
|
||||
print(event.delta.text, end="", flush=True)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Streaming with Tool Use
|
||||
|
||||
The Python tool runner supports streaming: pass `stream=True` to `client.beta.messages.tool_runner(...)` and each iteration yields a stream you consume event-by-event, with `get_final_message()` for the accumulated message per turn (see `shared/tool-use-concepts.md` → Tool Runner vs Manual Loop). Use the manual-loop pattern below only when you're not using the tool runner and need per-token streaming with tools:
|
||||
|
||||
```python
|
||||
with client.messages.stream(
|
||||
model="claude-opus-5",
|
||||
max_tokens=64000,
|
||||
tools=tools,
|
||||
messages=messages
|
||||
) as stream:
|
||||
for text in stream.text_stream:
|
||||
print(text, end="", flush=True)
|
||||
|
||||
response = stream.get_final_message()
|
||||
# Continue with tool execution if response.stop_reason == "tool_use"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Getting the Final Message
|
||||
|
||||
```python
|
||||
with client.messages.stream(
|
||||
model="claude-opus-5",
|
||||
max_tokens=64000,
|
||||
messages=[{"role": "user", "content": "Hello"}]
|
||||
) as stream:
|
||||
for text in stream.text_stream:
|
||||
print(text, end="", flush=True)
|
||||
|
||||
# Get full message after streaming
|
||||
final_message = stream.get_final_message()
|
||||
print(f"\n\nTokens used: {final_message.usage.output_tokens}")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Streaming with Progress Updates
|
||||
|
||||
```python
|
||||
def stream_with_progress(client, **kwargs):
|
||||
"""Stream a response with progress updates."""
|
||||
total_tokens = 0
|
||||
content_parts = []
|
||||
|
||||
with client.messages.stream(**kwargs) as stream:
|
||||
for event in stream:
|
||||
if event.type == "content_block_delta":
|
||||
if event.delta.type == "text_delta":
|
||||
text = event.delta.text
|
||||
content_parts.append(text)
|
||||
print(text, end="", flush=True)
|
||||
|
||||
elif event.type == "message_delta":
|
||||
if event.usage and event.usage.output_tokens is not None:
|
||||
total_tokens = event.usage.output_tokens
|
||||
|
||||
final_message = stream.get_final_message()
|
||||
|
||||
print(f"\n\n[Tokens used: {total_tokens}]")
|
||||
return "".join(content_parts)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Error Handling in Streams
|
||||
|
||||
```python
|
||||
try:
|
||||
with client.messages.stream(
|
||||
model="claude-opus-5",
|
||||
max_tokens=64000,
|
||||
messages=[{"role": "user", "content": "Write a story"}]
|
||||
) as stream:
|
||||
for text in stream.text_stream:
|
||||
print(text, end="", flush=True)
|
||||
except anthropic.APIConnectionError:
|
||||
print("\nConnection lost. Please retry.")
|
||||
except anthropic.RateLimitError:
|
||||
print("\nRate limited. Please wait and retry.")
|
||||
except anthropic.APIStatusError as e:
|
||||
print(f"\nAPI error: {e.status_code}")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Stream Event Types
|
||||
|
||||
| Event Type | Description | When it fires |
|
||||
| --------------------- | --------------------------- | --------------------------------- |
|
||||
| `message_start` | Contains message metadata | Once at the beginning |
|
||||
| `content_block_start` | New content block beginning | When a text/tool_use block starts |
|
||||
| `content_block_delta` | Incremental content update | For each token/chunk |
|
||||
| `content_block_stop` | Content block complete | When a block finishes |
|
||||
| `message_delta` | Message-level updates | Contains `stop_reason`, usage |
|
||||
| `message_stop` | Message complete | Once at the end |
|
||||
|
||||
## Best Practices
|
||||
|
||||
1. **Always flush output** — Use `flush=True` to show tokens immediately
|
||||
2. **Handle partial responses** — If the stream is interrupted, you may have incomplete content
|
||||
3. **Track token usage** — The `message_delta` event contains usage information
|
||||
4. **Use timeouts** — Set appropriate timeouts for your application
|
||||
5. **Default to streaming** — Use `.get_final_message()` to get the complete response even when streaming, giving you timeout protection without needing to handle individual events
|
||||
6. **Large `max_tokens` without streaming raises `ValueError`** — The SDK refuses non-streaming requests it estimates will exceed ~10 minutes (idle connections drop). Pass `stream=True` / use `messages.stream()`, or explicitly override `timeout`, to suppress the guard.
|
||||
629
.agents/skills/claude-api/python/claude-api/tool-use.md
Normal file
629
.agents/skills/claude-api/python/claude-api/tool-use.md
Normal file
@@ -0,0 +1,629 @@
|
||||
# Tool Use — Python
|
||||
|
||||
For conceptual overview (tool definitions, tool choice, tips), see [shared/tool-use-concepts.md](../../shared/tool-use-concepts.md).
|
||||
|
||||
## Tool Runner (Recommended)
|
||||
|
||||
**Beta:** The tool runner is in beta in the Python SDK.
|
||||
|
||||
Use the `@beta_tool` decorator to define tools as typed functions, then pass them to `client.beta.messages.tool_runner()`:
|
||||
|
||||
```python
|
||||
import anthropic
|
||||
from anthropic import beta_tool
|
||||
|
||||
client = anthropic.Anthropic()
|
||||
|
||||
@beta_tool
|
||||
def get_weather(location: str, unit: str = "celsius") -> str:
|
||||
"""Get current weather for a location.
|
||||
|
||||
Args:
|
||||
location: City and state, e.g., San Francisco, CA.
|
||||
unit: Temperature unit, either "celsius" or "fahrenheit".
|
||||
"""
|
||||
# Your implementation here
|
||||
return f"72°F and sunny in {location}"
|
||||
|
||||
# The tool runner handles the agentic loop automatically
|
||||
runner = client.beta.messages.tool_runner(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
tools=[get_weather],
|
||||
messages=[{"role": "user", "content": "What's the weather in Paris?"}],
|
||||
)
|
||||
|
||||
# Each iteration yields a BetaMessage; iteration stops when Claude is done
|
||||
for message in runner:
|
||||
print(message)
|
||||
```
|
||||
|
||||
For async usage, use `@beta_async_tool` with `async def` functions.
|
||||
|
||||
**Key benefits of the tool runner:**
|
||||
|
||||
- No manual loop — the SDK handles calling tools and feeding results back
|
||||
- Type-safe tool inputs via decorators
|
||||
- Tool schemas are generated automatically from function signatures
|
||||
- Iteration stops automatically when Claude has no more tool calls
|
||||
|
||||
### Server tools with the tool runner
|
||||
|
||||
The runner's `tools` list accepts raw server-tool definitions (`web_search_20260209`, `web_fetch_20260209`, code execution) alongside decorated tools — pass the literal tool dict; server tools run on Anthropic's servers, so there is no function to implement.
|
||||
|
||||
**Caution — the runner does not auto-resume `pause_turn` (as of `anthropic` 0.116.0).** A long-running server-tool turn can stop with `stop_reason: "pause_turn"`. The runner only continues after a client tool produces a result, so a paused turn ends the loop and is returned as the final message — no error, no warning, just a silently truncated answer. Unlike the TypeScript runner, the Python runner cannot be resumed mid-loop: it exits unconditionally when no client tool ran, and `runner.append_messages(...)` does not prevent the exit. To handle `pause_turn`, mirror the conversation history as you iterate, then restart the runner with the paused turn appended:
|
||||
|
||||
```python
|
||||
messages = [{"role": "user", "content": user_input}]
|
||||
|
||||
max_restarts = 5 # cap pause_turn restarts, mirroring max_continuations advice
|
||||
restarts = 0
|
||||
while True:
|
||||
runner = client.beta.messages.tool_runner(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
tools=tools, # may mix @beta_tool functions and server-tool definitions
|
||||
messages=messages,
|
||||
)
|
||||
last = None
|
||||
for message in runner:
|
||||
last = message
|
||||
# Mirror the history — the runner keeps its own copy and does not expose it
|
||||
messages.append({"role": "assistant", "content": message.content})
|
||||
tool_response = runner.generate_tool_call_response() # cached; tools still run once
|
||||
if tool_response is not None:
|
||||
messages.append(tool_response)
|
||||
if last is None or last.stop_reason != "pause_turn":
|
||||
break
|
||||
restarts += 1
|
||||
if restarts > max_restarts:
|
||||
raise RuntimeError("giving up: turn still paused after max_restarts")
|
||||
# Paused mid-turn: `messages` already ends with the paused assistant
|
||||
# turn, so the next runner resumes it
|
||||
```
|
||||
|
||||
Alternatively, use the manual loop below, which handles `pause_turn` explicitly.
|
||||
|
||||
---
|
||||
|
||||
## MCP Tool Conversion Helpers
|
||||
|
||||
**Beta.** Convert [MCP (Model Context Protocol)](https://modelcontextprotocol.io/) tools, prompts, and resources to Anthropic API types for use with the tool runner. Requires `pip install anthropic[mcp]` (Python 3.10+).
|
||||
|
||||
> **Note:** The Claude API also supports an `mcp_servers` parameter that lets Claude connect directly to remote MCP servers. Use these helpers instead when you need local MCP servers, prompts, resources, or more control over the MCP connection.
|
||||
|
||||
### MCP Tools with Tool Runner
|
||||
|
||||
```python
|
||||
from anthropic import AsyncAnthropic
|
||||
from anthropic.lib.tools.mcp import async_mcp_tool
|
||||
from mcp import ClientSession
|
||||
from mcp.client.stdio import stdio_client, StdioServerParameters
|
||||
|
||||
client = AsyncAnthropic()
|
||||
|
||||
async with stdio_client(StdioServerParameters(command="mcp-server")) as (read, write):
|
||||
async with ClientSession(read, write) as mcp_client:
|
||||
await mcp_client.initialize()
|
||||
|
||||
tools_result = await mcp_client.list_tools()
|
||||
# tool_runner is sync — returns the runner, not a coroutine
|
||||
runner = client.beta.messages.tool_runner(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{"role": "user", "content": "Use the available tools"}],
|
||||
tools=[async_mcp_tool(t, mcp_client) for t in tools_result.tools],
|
||||
)
|
||||
async for message in runner:
|
||||
print(message)
|
||||
```
|
||||
|
||||
For sync usage, use `mcp_tool` instead of `async_mcp_tool`.
|
||||
|
||||
### MCP Prompts
|
||||
|
||||
```python
|
||||
from anthropic.lib.tools.mcp import mcp_message
|
||||
|
||||
prompt = await mcp_client.get_prompt(name="my-prompt")
|
||||
response = await client.beta.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[mcp_message(m) for m in prompt.messages],
|
||||
)
|
||||
```
|
||||
|
||||
### MCP Resources as Content
|
||||
|
||||
```python
|
||||
from anthropic.lib.tools.mcp import mcp_resource_to_content
|
||||
|
||||
resource = await mcp_client.read_resource(uri="file:///path/to/doc.txt")
|
||||
response = await client.beta.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": [
|
||||
mcp_resource_to_content(resource),
|
||||
{"type": "text", "text": "Summarize this document"},
|
||||
],
|
||||
}],
|
||||
)
|
||||
```
|
||||
|
||||
### Upload MCP Resources as Files
|
||||
|
||||
```python
|
||||
from anthropic.lib.tools.mcp import mcp_resource_to_file
|
||||
|
||||
resource = await mcp_client.read_resource(uri="file:///path/to/data.json")
|
||||
uploaded = await client.beta.files.upload(file=mcp_resource_to_file(resource))
|
||||
```
|
||||
|
||||
Conversion functions raise `UnsupportedMCPValueError` if an MCP value cannot be converted (e.g., unsupported content types like audio, unsupported MIME types).
|
||||
|
||||
---
|
||||
|
||||
## Manual Agentic Loop
|
||||
|
||||
Prefer the tool runner above. Drop to a manual loop only when you need control the runner does not expose (e.g., a custom transport, request shapes the SDK cannot build, or avoiding a beta dependency — the runner is beta). Human-in-the-loop approval does *not* require a manual loop — gate inside the tool function (return a "user declined" result) or inspect pending `tool_use` blocks in the `for message in runner:` body and call `runner.set_messages_params()`.
|
||||
|
||||
If you do need a manual loop:
|
||||
|
||||
```python
|
||||
import anthropic
|
||||
|
||||
client = anthropic.Anthropic()
|
||||
tools = [...] # Your tool definitions
|
||||
messages = [{"role": "user", "content": user_input}]
|
||||
|
||||
# Agentic loop: keep going until Claude stops calling tools
|
||||
while True:
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
tools=tools,
|
||||
messages=messages
|
||||
)
|
||||
|
||||
# If Claude is done (no more tool calls), break
|
||||
if response.stop_reason == "end_turn":
|
||||
break
|
||||
|
||||
# Server-side tool hit iteration limit; re-send to continue
|
||||
if response.stop_reason == "pause_turn":
|
||||
messages = [
|
||||
{"role": "user", "content": user_input},
|
||||
{"role": "assistant", "content": response.content},
|
||||
]
|
||||
continue
|
||||
|
||||
# Extract tool use blocks from the response
|
||||
tool_use_blocks = [b for b in response.content if b.type == "tool_use"]
|
||||
|
||||
# Append assistant's response (including tool_use blocks)
|
||||
messages.append({"role": "assistant", "content": response.content})
|
||||
|
||||
# Execute each tool and collect results
|
||||
tool_results = []
|
||||
for tool in tool_use_blocks:
|
||||
result = execute_tool(tool.name, tool.input) # Your implementation
|
||||
tool_results.append({
|
||||
"type": "tool_result",
|
||||
"tool_use_id": tool.id, # Must match the tool_use block's id
|
||||
"content": result
|
||||
})
|
||||
|
||||
# Append tool results as a user message
|
||||
messages.append({"role": "user", "content": tool_results})
|
||||
|
||||
# Final response text
|
||||
final_text = next(b.text for b in response.content if b.type == "text")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Handling Tool Results
|
||||
|
||||
```python
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
tools=tools,
|
||||
messages=[{"role": "user", "content": "What's the weather in Paris?"}]
|
||||
)
|
||||
|
||||
for block in response.content:
|
||||
if block.type == "tool_use":
|
||||
tool_name = block.name
|
||||
tool_input = block.input
|
||||
tool_use_id = block.id
|
||||
|
||||
result = execute_tool(tool_name, tool_input)
|
||||
|
||||
followup = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
tools=tools,
|
||||
messages=[
|
||||
{"role": "user", "content": "What's the weather in Paris?"},
|
||||
{"role": "assistant", "content": response.content},
|
||||
{
|
||||
"role": "user",
|
||||
"content": [{
|
||||
"type": "tool_result",
|
||||
"tool_use_id": tool_use_id,
|
||||
"content": result
|
||||
}]
|
||||
}
|
||||
]
|
||||
)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Multiple Tool Calls
|
||||
|
||||
```python
|
||||
tool_results = []
|
||||
|
||||
for block in response.content:
|
||||
if block.type == "tool_use":
|
||||
result = execute_tool(block.name, block.input)
|
||||
tool_results.append({
|
||||
"type": "tool_result",
|
||||
"tool_use_id": block.id,
|
||||
"content": result
|
||||
})
|
||||
|
||||
# Send all results back at once
|
||||
if tool_results:
|
||||
followup = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
tools=tools,
|
||||
messages=[
|
||||
*previous_messages,
|
||||
{"role": "assistant", "content": response.content},
|
||||
{"role": "user", "content": tool_results}
|
||||
]
|
||||
)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Error Handling in Tool Results
|
||||
|
||||
```python
|
||||
tool_result = {
|
||||
"type": "tool_result",
|
||||
"tool_use_id": tool_use_id,
|
||||
"content": "Error: Location 'xyz' not found. Please provide a valid city name.",
|
||||
"is_error": True
|
||||
}
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Tool Choice
|
||||
|
||||
```python
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
tools=tools,
|
||||
tool_choice={"type": "tool", "name": "get_weather"}, # Force specific tool
|
||||
messages=[{"role": "user", "content": "What's the weather in Paris?"}]
|
||||
)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Code Execution
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
import anthropic
|
||||
|
||||
client = anthropic.Anthropic()
|
||||
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": "Calculate the mean and standard deviation of [1, 2, 3, 4, 5, 6, 7, 8, 9, 10]"
|
||||
}],
|
||||
tools=[{
|
||||
"type": "code_execution_20260120",
|
||||
"name": "code_execution"
|
||||
}]
|
||||
)
|
||||
|
||||
for block in response.content:
|
||||
if block.type == "text":
|
||||
print(block.text)
|
||||
elif block.type == "bash_code_execution_tool_result":
|
||||
print(f"stdout: {block.content.stdout}")
|
||||
```
|
||||
|
||||
### Upload Files for Analysis
|
||||
|
||||
```python
|
||||
# 1. Upload a file
|
||||
uploaded = client.beta.files.upload(file=open("sales_data.csv", "rb"))
|
||||
|
||||
# 2. Pass to code execution via container_upload block
|
||||
# Code execution is GA; Files API is still beta (pass via extra_headers)
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
extra_headers={"anthropic-beta": "files-api-2025-04-14"},
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": [
|
||||
{"type": "text", "text": "Analyze this sales data. Show trends and create a visualization."},
|
||||
{"type": "container_upload", "file_id": uploaded.id}
|
||||
]
|
||||
}],
|
||||
tools=[{"type": "code_execution_20260120", "name": "code_execution"}]
|
||||
)
|
||||
```
|
||||
|
||||
### Retrieve Generated Files
|
||||
|
||||
```python
|
||||
import os
|
||||
|
||||
OUTPUT_DIR = "./claude_outputs"
|
||||
os.makedirs(OUTPUT_DIR, exist_ok=True)
|
||||
|
||||
for block in response.content:
|
||||
if block.type == "bash_code_execution_tool_result":
|
||||
result = block.content
|
||||
if result.type == "bash_code_execution_result" and result.content:
|
||||
for file_ref in result.content:
|
||||
if file_ref.type == "bash_code_execution_output":
|
||||
metadata = client.beta.files.retrieve_metadata(file_ref.file_id)
|
||||
file_content = client.beta.files.download(file_ref.file_id)
|
||||
# Use basename to prevent path traversal; validate result
|
||||
safe_name = os.path.basename(metadata.filename)
|
||||
if not safe_name or safe_name in (".", ".."):
|
||||
print(f"Skipping invalid filename: {metadata.filename}")
|
||||
continue
|
||||
output_path = os.path.join(OUTPUT_DIR, safe_name)
|
||||
file_content.write_to_file(output_path)
|
||||
print(f"Saved: {output_path}")
|
||||
```
|
||||
|
||||
### Container Reuse
|
||||
|
||||
```python
|
||||
# First request: set up environment
|
||||
response1 = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{"role": "user", "content": "Install tabulate and create data.json with sample data"}],
|
||||
tools=[{"type": "code_execution_20260120", "name": "code_execution"}]
|
||||
)
|
||||
|
||||
# Get container ID from response
|
||||
container_id = response1.container.id
|
||||
|
||||
# Second request: reuse the same container
|
||||
response2 = client.messages.create(
|
||||
container=container_id,
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{"role": "user", "content": "Read data.json and display as a formatted table"}],
|
||||
tools=[{"type": "code_execution_20260120", "name": "code_execution"}]
|
||||
)
|
||||
```
|
||||
|
||||
### Response Structure
|
||||
|
||||
```python
|
||||
for block in response.content:
|
||||
if block.type == "text":
|
||||
print(block.text) # Claude's explanation
|
||||
elif block.type == "server_tool_use":
|
||||
print(f"Running: {block.name} - {block.input}") # What Claude is doing
|
||||
elif block.type == "bash_code_execution_tool_result":
|
||||
result = block.content
|
||||
if result.type == "bash_code_execution_result":
|
||||
if result.return_code == 0:
|
||||
print(f"Output: {result.stdout}")
|
||||
else:
|
||||
print(f"Error: {result.stderr}")
|
||||
else:
|
||||
print(f"Tool error: {result.error_code}")
|
||||
elif block.type == "text_editor_code_execution_tool_result":
|
||||
print(f"File operation: {block.content}")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Memory Tool
|
||||
|
||||
### Basic Usage
|
||||
|
||||
```python
|
||||
import anthropic
|
||||
|
||||
client = anthropic.Anthropic()
|
||||
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{"role": "user", "content": "Remember that my preferred language is Python."}],
|
||||
tools=[{"type": "memory_20250818", "name": "memory"}],
|
||||
)
|
||||
```
|
||||
|
||||
### SDK Memory Helper
|
||||
|
||||
Subclass `BetaAbstractMemoryTool`:
|
||||
|
||||
```python
|
||||
from anthropic.lib.tools import BetaAbstractMemoryTool
|
||||
|
||||
class MyMemoryTool(BetaAbstractMemoryTool):
|
||||
def view(self, command): ...
|
||||
def create(self, command): ...
|
||||
def str_replace(self, command): ...
|
||||
def insert(self, command): ...
|
||||
def delete(self, command): ...
|
||||
def rename(self, command): ...
|
||||
|
||||
memory = MyMemoryTool()
|
||||
|
||||
# Use with tool runner
|
||||
runner = client.beta.messages.tool_runner(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
tools=[memory],
|
||||
messages=[{"role": "user", "content": "Remember my preferences"}],
|
||||
)
|
||||
|
||||
for message in runner:
|
||||
print(message)
|
||||
```
|
||||
|
||||
For full implementation examples, use WebFetch:
|
||||
|
||||
- `https://github.com/anthropics/anthropic-sdk-python/blob/main/examples/memory/basic.py`
|
||||
|
||||
---
|
||||
|
||||
## Structured Outputs
|
||||
|
||||
### JSON Outputs (Pydantic — Recommended)
|
||||
|
||||
```python
|
||||
from pydantic import BaseModel
|
||||
from typing import List
|
||||
import anthropic
|
||||
|
||||
class ContactInfo(BaseModel):
|
||||
name: str
|
||||
email: str
|
||||
plan: str
|
||||
interests: List[str]
|
||||
demo_requested: bool
|
||||
|
||||
client = anthropic.Anthropic()
|
||||
|
||||
response = client.messages.parse(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": "Extract: Jane Doe (jane@co.com) wants Enterprise, interested in API and SDKs, wants a demo."
|
||||
}],
|
||||
output_format=ContactInfo,
|
||||
)
|
||||
|
||||
# response.parsed_output is a validated ContactInfo instance
|
||||
contact = response.parsed_output
|
||||
print(contact.name) # "Jane Doe"
|
||||
print(contact.interests) # ["API", "SDKs"]
|
||||
```
|
||||
|
||||
### Raw Schema
|
||||
|
||||
```python
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{
|
||||
"role": "user",
|
||||
"content": "Extract info: John Smith (john@example.com) wants the Enterprise plan."
|
||||
}],
|
||||
output_config={
|
||||
"format": {
|
||||
"type": "json_schema",
|
||||
"schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"name": {"type": "string"},
|
||||
"email": {"type": "string"},
|
||||
"plan": {"type": "string"},
|
||||
"demo_requested": {"type": "boolean"}
|
||||
},
|
||||
"required": ["name", "email", "plan", "demo_requested"],
|
||||
"additionalProperties": False
|
||||
}
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
import json
|
||||
# output_config.format guarantees the first block is text with valid JSON
|
||||
text = next(b.text for b in response.content if b.type == "text")
|
||||
data = json.loads(text)
|
||||
```
|
||||
|
||||
### Strict Tool Use
|
||||
|
||||
```python
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{"role": "user", "content": "Book a flight to Tokyo for 2 passengers on March 15"}],
|
||||
tools=[{
|
||||
"name": "book_flight",
|
||||
"description": "Book a flight to a destination",
|
||||
"strict": True,
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"destination": {"type": "string"},
|
||||
"date": {"type": "string", "format": "date"},
|
||||
"passengers": {"type": "integer", "enum": [1, 2, 3, 4, 5, 6, 7, 8]}
|
||||
},
|
||||
"required": ["destination", "date", "passengers"],
|
||||
"additionalProperties": False
|
||||
}
|
||||
}]
|
||||
)
|
||||
```
|
||||
|
||||
### Using Both Together
|
||||
|
||||
```python
|
||||
response = client.messages.create(
|
||||
model="claude-opus-5",
|
||||
max_tokens=16000,
|
||||
messages=[{"role": "user", "content": "Plan a trip to Paris next month"}],
|
||||
output_config={
|
||||
"format": {
|
||||
"type": "json_schema",
|
||||
"schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"summary": {"type": "string"},
|
||||
"next_steps": {"type": "array", "items": {"type": "string"}}
|
||||
},
|
||||
"required": ["summary", "next_steps"],
|
||||
"additionalProperties": False
|
||||
}
|
||||
}
|
||||
},
|
||||
tools=[{
|
||||
"name": "search_flights",
|
||||
"description": "Search for available flights",
|
||||
"strict": True,
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"destination": {"type": "string"},
|
||||
"date": {"type": "string", "format": "date"}
|
||||
},
|
||||
"required": ["destination", "date"],
|
||||
"additionalProperties": False
|
||||
}
|
||||
}]
|
||||
)
|
||||
```
|
||||
335
.agents/skills/claude-api/python/managed-agents/README.md
Normal file
335
.agents/skills/claude-api/python/managed-agents/README.md
Normal file
@@ -0,0 +1,335 @@
|
||||
# Managed Agents — Python
|
||||
|
||||
> **Bindings not shown here:** This README covers the most common managed-agents flows for Python. If you need a class, method, namespace, field, or behavior that isn't shown, WebFetch the Python SDK repo **or the relevant docs page** from `shared/live-sources.md` rather than guess. Do not extrapolate from cURL shapes or another language's SDK.
|
||||
|
||||
> **Agents are persistent — create once, reference by ID.** Store the agent ID returned by `agents.create` and pass it to every subsequent `sessions.create`; do not call `agents.create` in the request path. **Recommended:** define agents and environments as version-controlled YAML applied with the `ant` CLI — see `shared/anthropic-cli.md` (its live-docs URL is in `shared/live-sources.md`). The CLI owns the control plane (create/update); your code owns the data plane (sessions with the stored ID). The examples below show in-code creation for when you must provision programmatically; in production the create call belongs in setup, not in the request path.
|
||||
|
||||
## Installation
|
||||
|
||||
```bash
|
||||
pip install anthropic
|
||||
```
|
||||
|
||||
## Client Initialization
|
||||
|
||||
```python
|
||||
import anthropic
|
||||
|
||||
# Default — resolves credentials from the environment:
|
||||
# ANTHROPIC_API_KEY, or ANTHROPIC_AUTH_TOKEN, or an `ant auth login` profile.
|
||||
# Prefer this for local dev; don't hardcode a key.
|
||||
client = anthropic.Anthropic()
|
||||
|
||||
# Explicit API key (only when you must inject a specific key)
|
||||
client = anthropic.Anthropic(api_key="your-api-key")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Create an Environment
|
||||
|
||||
```python
|
||||
environment = client.beta.environments.create(
|
||||
name="my-dev-env",
|
||||
config={
|
||||
"type": "cloud",
|
||||
"networking": {"type": "unrestricted"},
|
||||
},
|
||||
)
|
||||
print(environment.id) # env_...
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Create an Agent (required first step)
|
||||
|
||||
> ⚠️ **There is no inline agent config.** `model`/`system`/`tools` live on the agent object, not the session. Always start with `agents.create()` — the session only takes `agent={"type": "agent", "id": agent.id}`.
|
||||
|
||||
### Minimal
|
||||
|
||||
```python
|
||||
# 1. Create the agent (reusable, versioned)
|
||||
agent = client.beta.agents.create(
|
||||
name="Coding Assistant",
|
||||
model="claude-opus-5",
|
||||
tools=[{"type": "agent_toolset_20260401", "default_config": {"enabled": True}}],
|
||||
)
|
||||
|
||||
# 2. Start a session
|
||||
session = client.beta.sessions.create(
|
||||
agent={"type": "agent", "id": agent.id, "version": agent.version},
|
||||
environment_id=environment.id,
|
||||
)
|
||||
print(session.id, session.status)
|
||||
print(f"Trace: https://platform.claude.com/workspaces/default/sessions/{session.id}") # swap 'default' for your workspace ID if the API key is not in the Default workspace
|
||||
```
|
||||
|
||||
### With system prompt and custom tools
|
||||
|
||||
```python
|
||||
import os
|
||||
|
||||
agent = client.beta.agents.create(
|
||||
name="Code Reviewer",
|
||||
model="claude-opus-5",
|
||||
system="You are a senior code reviewer.",
|
||||
tools=[
|
||||
{"type": "agent_toolset_20260401"},
|
||||
{
|
||||
"type": "custom",
|
||||
"name": "run_tests",
|
||||
"description": "Run the test suite",
|
||||
"input_schema": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"test_path": {"type": "string", "description": "Path to test file"}
|
||||
},
|
||||
"required": ["test_path"],
|
||||
},
|
||||
},
|
||||
],
|
||||
)
|
||||
|
||||
session = client.beta.sessions.create(
|
||||
agent={"type": "agent", "id": agent.id, "version": agent.version},
|
||||
environment_id=environment.id,
|
||||
title="Code review session",
|
||||
resources=[
|
||||
{
|
||||
"type": "github_repository",
|
||||
"url": "https://github.com/owner/repo",
|
||||
"mount_path": "/workspace/repo",
|
||||
"authorization_token": os.environ["GITHUB_TOKEN"],
|
||||
"branch": "main",
|
||||
}
|
||||
],
|
||||
)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Send a User Message
|
||||
|
||||
```python
|
||||
client.beta.sessions.events.send(
|
||||
session_id=session.id,
|
||||
events=[
|
||||
{
|
||||
"type": "user.message",
|
||||
"content": [{"type": "text", "text": "Review the auth module"}],
|
||||
}
|
||||
],
|
||||
)
|
||||
```
|
||||
|
||||
> 💡 **Stream-first:** Open the stream *before* (or concurrently with) sending the message. The stream only delivers events that occur after it opens — stream-after-send means early events arrive buffered in one batch. See [Steering Patterns](../../shared/managed-agents-events.md#steering-patterns).
|
||||
|
||||
---
|
||||
|
||||
## Stream Events (SSE)
|
||||
|
||||
```python
|
||||
import json
|
||||
|
||||
# Stream-first: open stream, then send while stream is live
|
||||
with client.beta.sessions.events.stream(
|
||||
session_id=session.id,
|
||||
) as stream:
|
||||
client.beta.sessions.events.send(
|
||||
session_id=session.id,
|
||||
events=[{"type": "user.message", "content": [{"type": "text", "text": "..."}]}],
|
||||
)
|
||||
for event in stream:
|
||||
... # process events
|
||||
|
||||
# Standalone stream iteration:
|
||||
with client.beta.sessions.events.stream(
|
||||
session_id=session.id,
|
||||
) as stream:
|
||||
for event in stream:
|
||||
if event.type == "agent.message":
|
||||
for block in event.content:
|
||||
if block.type == "text":
|
||||
print(block.text, end="", flush=True)
|
||||
elif event.type == "agent.custom_tool_use":
|
||||
# Custom tool invocation — session is now idle
|
||||
print(f"\nCustom tool call: {event.name}")
|
||||
print(f"Input: {json.dumps(event.input)}")
|
||||
# Send result back (see below)
|
||||
elif event.type == "session.status_idle":
|
||||
print("\n--- Agent idle ---")
|
||||
elif event.type == "session.status_terminated":
|
||||
print("\n--- Session terminated ---")
|
||||
break
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Provide Custom Tool Result
|
||||
|
||||
```python
|
||||
client.beta.sessions.events.send(
|
||||
session_id=session.id,
|
||||
events=[
|
||||
{
|
||||
"type": "user.custom_tool_result",
|
||||
"custom_tool_use_id": "sevt_abc123",
|
||||
"content": [{"type": "text", "text": "All 42 tests passed."}],
|
||||
}
|
||||
],
|
||||
)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Poll Events
|
||||
|
||||
```python
|
||||
events = client.beta.sessions.events.list(
|
||||
session_id=session.id,
|
||||
)
|
||||
for event in events.data:
|
||||
print(f"{event.type}: {event.id}")
|
||||
```
|
||||
|
||||
> ⚠️ **Prefer the SDK over raw `requests`/`httpx`.** If you hand-roll a poll loop, don't assume `timeout=(5, 60)` or `httpx.Timeout(120)` caps total call duration — both are **per-chunk** read timeouts (reset on every byte), so a trickling response can block forever. For a hard wall-clock deadline, track `time.monotonic()` at the loop level and bail explicitly, or wrap with `asyncio.wait_for()`. See [Receiving Events](../../shared/managed-agents-events.md#receiving-events).
|
||||
|
||||
---
|
||||
|
||||
## Full Streaming Loop with Custom Tools
|
||||
|
||||
```python
|
||||
import json
|
||||
|
||||
|
||||
def run_custom_tool(tool_name: str, tool_input: dict) -> str:
|
||||
"""Execute a custom tool and return the result."""
|
||||
if tool_name == "run_tests":
|
||||
# Your tool implementation here
|
||||
return "All tests passed."
|
||||
return f"Unknown tool: {tool_name}"
|
||||
|
||||
|
||||
def run_session(client, session_id: str):
|
||||
"""Stream events and handle custom tool calls."""
|
||||
while True:
|
||||
with client.beta.sessions.events.stream(
|
||||
session_id=session_id,
|
||||
) as stream:
|
||||
tool_calls = []
|
||||
for event in stream:
|
||||
if event.type == "agent.message":
|
||||
for block in event.content:
|
||||
if block.type == "text":
|
||||
print(block.text, end="", flush=True)
|
||||
elif event.type == "agent.custom_tool_use":
|
||||
tool_calls.append(event)
|
||||
elif event.type == "session.status_idle":
|
||||
break
|
||||
elif event.type == "session.status_terminated":
|
||||
return
|
||||
|
||||
if not tool_calls:
|
||||
break
|
||||
|
||||
# Process custom tool calls
|
||||
results = []
|
||||
for call in tool_calls:
|
||||
result = run_custom_tool(call.name, call.input)
|
||||
results.append({
|
||||
"type": "user.custom_tool_result",
|
||||
"custom_tool_use_id": call.id,
|
||||
"content": [{"type": "text", "text": result}],
|
||||
})
|
||||
|
||||
client.beta.sessions.events.send(
|
||||
session_id=session_id,
|
||||
events=results,
|
||||
)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Upload a File
|
||||
|
||||
```python
|
||||
with open("data.csv", "rb") as f:
|
||||
file = client.beta.files.upload(
|
||||
file=f,
|
||||
)
|
||||
|
||||
# Use in a session
|
||||
session = client.beta.sessions.create(
|
||||
agent={"type": "agent", "id": agent.id, "version": agent.version},
|
||||
environment_id=environment.id,
|
||||
resources=[{"type": "file", "file_id": file.id, "mount_path": "/workspace/data.csv"}],
|
||||
)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## List and Download Session Files
|
||||
|
||||
List files the agent wrote to `/mnt/session/outputs/` during a session, then download them.
|
||||
|
||||
```python
|
||||
# List files associated with a session
|
||||
files = client.beta.files.list(
|
||||
scope_id=session.id,
|
||||
betas=["managed-agents-2026-04-01"],
|
||||
)
|
||||
for f in files.data:
|
||||
print(f.filename, f.size_bytes)
|
||||
# Download each file and save to disk
|
||||
file_content = client.beta.files.download(f.id)
|
||||
file_content.write_to_file(f.filename)
|
||||
```
|
||||
|
||||
> 💡 There's a brief indexing lag (~1–3s) between `session.status_idle` and output files appearing in `files.list`. Retry once or twice if the list is empty.
|
||||
|
||||
---
|
||||
|
||||
## Session Management
|
||||
|
||||
```python
|
||||
# Get session details
|
||||
session = client.beta.sessions.retrieve(session_id="sesn_011CZxAbc123Def456")
|
||||
print(session.status, session.usage)
|
||||
|
||||
# List sessions
|
||||
sessions = client.beta.sessions.list()
|
||||
|
||||
# Delete a session
|
||||
client.beta.sessions.delete(session_id="sesn_011CZxAbc123Def456")
|
||||
|
||||
# Archive a session
|
||||
client.beta.sessions.archive(session_id="sesn_011CZxAbc123Def456")
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## MCP Server Integration
|
||||
|
||||
```python
|
||||
# Agent declares MCP server (no auth here — auth goes in a vault)
|
||||
agent = client.beta.agents.create(
|
||||
name="MCP Agent",
|
||||
model="claude-opus-5",
|
||||
mcp_servers=[
|
||||
{"type": "url", "name": "my-tools", "url": "https://my-mcp-server.example.com/sse"},
|
||||
],
|
||||
tools=[
|
||||
{"type": "agent_toolset_20260401", "default_config": {"enabled": True}},
|
||||
{"type": "mcp_toolset", "mcp_server_name": "my-tools"},
|
||||
],
|
||||
)
|
||||
|
||||
# Session attaches vault(s) containing credentials for those MCP server URLs
|
||||
session = client.beta.sessions.create(
|
||||
agent=agent.id,
|
||||
environment_id=environment.id,
|
||||
vault_ids=[vault.id],
|
||||
)
|
||||
```
|
||||
|
||||
See `shared/managed-agents-tools.md` §Vaults for creating vaults and adding credentials.
|
||||
Reference in New Issue
Block a user