Compare commits
216
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fa66e00547 | ||
|
|
59b4abaa14 | ||
|
|
71e65052d1 | ||
|
|
7b0714c5c5 | ||
|
|
4bdcd0b568 | ||
|
|
fdecb76035 | ||
|
|
b7d451ec5d | ||
|
|
86fe3a4749 | ||
|
|
76d5a73cc7 | ||
|
|
2ab6494ec9 | ||
|
|
3f2684dcfe | ||
|
|
266458528e | ||
|
|
35eb35cdc2 | ||
|
|
8cb5d93005 | ||
|
|
5569c99b8e | ||
|
|
d90c3b4a24 | ||
|
|
ee0b25e29a | ||
|
|
a3fe901886 | ||
|
|
153b08f872 | ||
|
|
1b920d7299 | ||
|
|
4b3c42ad5c | ||
|
|
0de186071b | ||
|
|
7bcd6c5349 | ||
|
|
08b399a450 | ||
|
|
97d5bd3c4d | ||
|
|
a8f408b3b0 | ||
|
|
0bdb762832 | ||
|
|
d49e009b12 | ||
|
|
7dc400c05c | ||
|
|
65aca4d260 | ||
|
|
34584c3a2e | ||
|
|
53e09b924c | ||
|
|
1b302ab4bf | ||
|
|
3c681f1639 | ||
|
|
1ff3356d1b | ||
|
|
5193e34803 | ||
|
|
8f8fc81135 | ||
|
|
d4abb3d06f | ||
|
|
b2570f1a62 | ||
|
|
f19b5f5929 | ||
|
|
8e829396b2 | ||
|
|
e8e8ca6700 | ||
|
|
f1cbd4d730 | ||
|
|
f7cebfe7f3 | ||
|
|
b854d9a888 | ||
|
|
83d2acf07f | ||
|
|
eee9c38953 | ||
|
|
e782318338 | ||
|
|
dc94aa76cc | ||
|
|
5cf019c21e | ||
|
|
790bdd6b8a | ||
|
|
b25c09f5ed | ||
|
|
9e8c910ab1 | ||
|
|
cc10e20a47 | ||
|
|
34ed4345fc | ||
|
|
1a85333e4c | ||
|
|
3c587c788a | ||
|
|
303d123527 | ||
|
|
61c2cb4ac4 | ||
|
|
3126b99fdb | ||
|
|
b28b647ce3 | ||
|
|
d736f1cf46 | ||
|
|
e4402f2f83 | ||
|
|
dc5d8edfec | ||
|
|
bdc3be650b | ||
|
|
119de1f347 | ||
|
|
88f6ecea5a | ||
|
|
a0eb6e9dcf | ||
|
|
c987976f82 | ||
|
|
8116848670 | ||
|
|
f6412b8349 | ||
|
|
f1023d9573 | ||
|
|
39560524f7 | ||
|
|
aec8510d49 | ||
|
|
9dd9c0a4be | ||
|
|
53391762be | ||
|
|
d2487ec6a3 | ||
|
|
8e7c6db4b5 | ||
|
|
a16020c4a2 | ||
|
|
83d6e3cd65 | ||
|
|
80e56294f6 | ||
|
|
a1823004aa | ||
|
|
54255c89c4 | ||
|
|
9a7596193f | ||
|
|
31a889f9fd | ||
|
|
6c0d68cfbb | ||
|
|
5234e76c40 | ||
|
|
f048e8cbec | ||
|
|
9e8d2d4a09 | ||
|
|
aa4393b2eb | ||
|
|
9fcb1dc4c0 | ||
|
|
c5f78bf17e | ||
|
|
6003777eda | ||
|
|
bbfb4a0c2c | ||
|
|
b44c94c9cf | ||
|
|
0e16f1bc3d | ||
|
|
83fc393359 | ||
|
|
2e63a09150 | ||
|
|
16791c9717 | ||
|
|
a770d8b0f9 | ||
|
|
f148ffd7a4 | ||
|
|
03d3e3a4da | ||
|
|
6c13a7e722 | ||
|
|
6adefbf190 | ||
|
|
45a377030f | ||
|
|
6165f523c3 | ||
|
|
3c760be0b2 | ||
|
|
b322cd66f2 | ||
|
|
9bb9cd0d55 | ||
|
|
ae1dc44705 | ||
|
|
548179ed3b | ||
|
|
88828112b5 | ||
|
|
e9a221bdd9 | ||
|
|
003c8dbf59 | ||
|
|
51b5e02948 | ||
|
|
97ad118615 | ||
|
|
5d7d526ca0 | ||
|
|
d987d04606 | ||
|
|
e99a47608d | ||
|
|
d584c11b6f | ||
|
|
5de628434c | ||
|
|
fd8365992d | ||
|
|
0fb9504920 | ||
|
|
6e868cb712 | ||
|
|
0ca40f1929 | ||
|
|
195483c65f | ||
|
|
e140b850b4 | ||
|
|
b4c6c4e5ef | ||
|
|
b0e2033ded | ||
|
|
d095bd3cb8 | ||
|
|
5ef45c4345 | ||
|
|
215637a93c | ||
|
|
c6b68f0b6b | ||
|
|
5317bf869b | ||
|
|
0ea1af4ebf | ||
|
|
ca56d15dc9 | ||
|
|
51f38af9eb | ||
|
|
dbd4786b49 | ||
|
|
a2788023a1 | ||
|
|
ba5863f34c | ||
|
|
513582720a | ||
|
|
8e7e94e424 | ||
|
|
31eae748f6 | ||
|
|
ad2d5d2e8f | ||
|
|
a27220dbd0 | ||
|
|
326f18f8a8 | ||
|
|
6f2ff279ae | ||
|
|
41a3366f3e | ||
|
|
5eba972737 | ||
|
|
dafaa3bab4 | ||
|
|
2059acb3a4 | ||
|
|
a8a075600e | ||
|
|
471fd08fba | ||
|
|
1d30c3f6ce | ||
|
|
f959185bca | ||
|
|
1381735e3b | ||
|
|
6612576f8f | ||
|
|
727ffa2943 | ||
|
|
8dc66c713a | ||
|
|
ca8376c4a6 | ||
|
|
e1987c7fa5 | ||
|
|
a267110ce3 | ||
|
|
c00979a3b8 | ||
|
|
171c18eb5a | ||
|
|
5924017c39 | ||
|
|
c34dd1c90f | ||
|
|
891b85403d | ||
|
|
ecf1029b08 | ||
|
|
45294c9cc6 | ||
|
|
c0d7107c07 | ||
|
|
4bec87c1e7 | ||
|
|
422cccf091 | ||
|
|
1579b6e052 | ||
|
|
9e3f4ad5cc | ||
|
|
001cec88c6 | ||
|
|
9854edf113 | ||
|
|
3b08980b9d | ||
|
|
75e840e831 | ||
|
|
95982b62ef | ||
|
|
6bf81753cc | ||
|
|
18482c72e4 | ||
|
|
19a81e1b05 | ||
|
|
73539ef1b8 | ||
|
|
47954fe260 | ||
|
|
38a1c7f838 | ||
|
|
bd03c151da | ||
|
|
8bbe412848 | ||
|
|
d26582c5dc | ||
|
|
b62de7799c | ||
|
|
82ca557b47 | ||
|
|
7fd1017a53 | ||
|
|
c4e78f4d80 | ||
|
|
05f4464935 | ||
|
|
ca9da38a92 | ||
|
|
a541817054 | ||
|
|
37e478a3e2 | ||
|
|
4e94e0a422 | ||
|
|
2d8a5de7b9 | ||
|
|
d6aaff511e | ||
|
|
fe74da53e2 | ||
|
|
dfb81b45a7 | ||
|
|
22629634c9 | ||
|
|
b507630f3b | ||
|
|
68a9b7ad7d | ||
|
|
536ede12de | ||
|
|
6fed0e5e2d | ||
|
|
283a1fbefd | ||
|
|
5b1bc3a47d | ||
|
|
aaf4de23cc | ||
|
|
f639364d7f | ||
|
|
4b0fb7bdbe | ||
|
|
5111c69ff9 | ||
|
|
12e1470506 | ||
|
|
8cb5dd28e6 | ||
|
|
44dc549f75 | ||
|
|
ee2bf70f42 |
@@ -0,0 +1,83 @@
|
||||
name: Build Nanobot OAuth
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: ['main']
|
||||
pull_request:
|
||||
branches: ['main']
|
||||
schedule:
|
||||
- cron: '0 3 * * *'
|
||||
workflow_dispatch:
|
||||
|
||||
env:
|
||||
REGISTRY: git.wylab.me
|
||||
IMAGE_NAME: wylab/nanobot
|
||||
BUILDKIT_PROGRESS: plain
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: [self-hosted, linux-amd64]
|
||||
timeout-minutes: 15
|
||||
permissions:
|
||||
contents: read
|
||||
packages: write
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v3
|
||||
|
||||
- name: Log in to the container registry
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: ${{ env.REGISTRY }}
|
||||
username: ${{ secrets.REGISTRY_USERNAME || github.actor }}
|
||||
password: ${{ secrets.REGISTRY_PASSWORD || secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Build and push Docker image
|
||||
uses: docker/build-push-action@v5
|
||||
with:
|
||||
context: .
|
||||
file: Dockerfile.oauth
|
||||
provenance: false
|
||||
platforms: linux/amd64
|
||||
cache-from: type=registry,ref=${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}:buildcache
|
||||
cache-to: type=registry,ref=${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}:buildcache,mode=max
|
||||
push: ${{ github.event_name != 'pull_request' }}
|
||||
tags: |
|
||||
${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}:latest
|
||||
${{ env.REGISTRY }}/${{ env.IMAGE_NAME }}:${{ github.sha }}
|
||||
|
||||
cleanup:
|
||||
if: github.event_name == 'push' || github.event_name == 'schedule'
|
||||
runs-on: [self-hosted, linux-amd64]
|
||||
needs: build
|
||||
steps:
|
||||
- name: Delete images older than 24h
|
||||
env:
|
||||
TOKEN: ${{ secrets.REGISTRY_PASSWORD || secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
cutoff=$(date -u -d '24 hours ago' +%s)
|
||||
page=1
|
||||
while true; do
|
||||
versions=$(curl -sf -H "Authorization: token $TOKEN" \
|
||||
"https://${{ env.REGISTRY }}/api/v1/packages/wylab?type=container&q=nanobot&limit=50&page=$page")
|
||||
count=$(echo "$versions" | jq length)
|
||||
[ "$count" = "0" ] && break
|
||||
echo "$versions" | jq -c '.[]' | while read -r pkg; do
|
||||
ver=$(echo "$pkg" | jq -r '.version')
|
||||
# Keep latest and buildcache, only delete SHA tags
|
||||
case "$ver" in latest|buildcache) continue ;; esac
|
||||
created=$(echo "$pkg" | jq -r '.created_at')
|
||||
ts=$(date -u -d "$created" +%s 2>/dev/null || echo 0)
|
||||
if [ "$ts" -lt "$cutoff" ]; then
|
||||
id=$(echo "$pkg" | jq -r '.id')
|
||||
echo "Deleting nanobot:$ver (id=$id, created=$created)"
|
||||
curl -sf -X DELETE -H "Authorization: token $TOKEN" \
|
||||
"https://${{ env.REGISTRY }}/api/v1/packages/wylab/container/nanobot/$ver" || true
|
||||
fi
|
||||
done
|
||||
[ "$count" -lt 50 ] && break
|
||||
page=$((page + 1))
|
||||
done
|
||||
@@ -15,6 +15,7 @@ docs/
|
||||
*.pyzz
|
||||
.venv/
|
||||
venv/
|
||||
.worktrees/
|
||||
__pycache__/
|
||||
poetry.lock
|
||||
.pytest_cache/
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
FROM birdxs/nanobot:latest
|
||||
|
||||
# ── Skill dependencies ──────────────────────────────────────────────
|
||||
|
||||
# APT: ffmpeg (video-frames, whisper), jq, tmux, build-essential (for go)
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ffmpeg jq tmux build-essential procps && rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# gh CLI via GitHub official apt repo
|
||||
RUN curl -fsSL https://cli.github.com/packages/githubcli-archive-keyring.gpg \
|
||||
| dd of=/usr/share/keyrings/githubcli-archive-keyring.gpg && \
|
||||
echo "deb [arch=amd64 signed-by=/usr/share/keyrings/githubcli-archive-keyring.gpg] https://cli.github.com/packages stable main" \
|
||||
> /etc/apt/sources.list.d/github-cli.list && \
|
||||
apt-get update && apt-get install -y gh && rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# Go toolchain
|
||||
RUN curl -fsSL https://go.dev/dl/go1.23.6.linux-amd64.tar.gz | tar -C /usr/local -xzf -
|
||||
ENV PATH="/usr/local/go/bin:/root/go/bin:${PATH}"
|
||||
|
||||
# Go tools: blogwatcher, blu (blucli), gifgrep, sonos (sonoscli), wacli, songsee
|
||||
RUN go install github.com/Hyaxia/blogwatcher/cmd/blogwatcher@latest && \
|
||||
go install github.com/steipete/blucli/cmd/blu@latest && \
|
||||
go install github.com/steipete/gifgrep/cmd/gifgrep@latest && \
|
||||
go install github.com/steipete/sonoscli/cmd/sonos@latest && \
|
||||
go install github.com/steipete/wacli/cmd/wacli@latest && \
|
||||
go install github.com/steipete/songsee/cmd/songsee@latest
|
||||
|
||||
# Pre-built binaries from GitHub releases
|
||||
# gogcli (gog)
|
||||
RUN curl -fsSL https://github.com/steipete/gogcli/releases/download/v0.9.0/gogcli_0.9.0_linux_amd64.tar.gz \
|
||||
| tar -xzf - -C /usr/local/bin gog
|
||||
|
||||
# goplaces
|
||||
RUN curl -fsSL https://github.com/steipete/goplaces/releases/download/v0.2.1/goplaces_0.2.1_linux_amd64.tar.gz \
|
||||
| tar -xzf - -C /usr/local/bin goplaces
|
||||
|
||||
# himalaya (email CLI)
|
||||
RUN curl -fsSL https://github.com/pimalaya/himalaya/releases/download/v1.1.0/himalaya.x86_64-linux.tgz \
|
||||
| tar -xzf - -C /usr/local/bin himalaya
|
||||
|
||||
# obsidian-cli (release binary is named notesmd-cli, skill expects obsidian-cli)
|
||||
RUN curl -fsSL -o /tmp/obsidian.tar.gz https://github.com/yakitrak/obsidian-cli/releases/download/v0.3.0/notesmd-cli_0.3.0_linux_amd64.tar.gz && \
|
||||
tar -xzf /tmp/obsidian.tar.gz -C /tmp notesmd-cli && \
|
||||
mv /tmp/notesmd-cli /usr/local/bin/obsidian-cli && \
|
||||
rm /tmp/obsidian.tar.gz
|
||||
|
||||
# Node tools: oracle, gemini-cli, summarize
|
||||
RUN npm install -g @steipete/oracle @google/gemini-cli @steipete/summarize
|
||||
|
||||
# Python tools: nano-pdf, openai-whisper
|
||||
RUN uv tool install nano-pdf && \
|
||||
uv tool install openai-whisper
|
||||
ENV PATH="/root/.local/bin:${PATH}"
|
||||
|
||||
# ── Nanobot source ──────────────────────────────────────────────────
|
||||
|
||||
COPY pyproject.toml README.md LICENSE /app/
|
||||
COPY nanobot/ /app/nanobot/
|
||||
RUN uv pip install --system --no-cache --reinstall /app[mem0] psycopg2-binary
|
||||
|
||||
ENTRYPOINT ["nanobot"]
|
||||
CMD ["gateway"]
|
||||
@@ -143,7 +143,7 @@ Add or merge these **two parts** into your config (other options have defaults).
|
||||
{
|
||||
"agents": {
|
||||
"defaults": {
|
||||
"model": "anthropic/claude-opus-4-5",
|
||||
"model": "anthropic/claude-opus-4-7",
|
||||
"provider": "openrouter"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,265 @@
|
||||
# Design: Native Anthropic Tools Integration
|
||||
|
||||
**Goal**: Integrate Anthropic's native trained tools (bash_20250124, text_editor_20250728, computer_20251124) into nanobot to leverage model's trained behaviors instead of custom function tools.
|
||||
|
||||
## Overview
|
||||
|
||||
Anthropic's native tools are version-coupled to model training. Unlike custom function tools (which the model learns via instruction-following at inference time), native tools have their behaviors baked into model weights during training. This provides more reliable tool execution.
|
||||
|
||||
**Key Insight**: The Anthropic API accepts BOTH tool formats in the same request:
|
||||
- Function tools: `{type: "function", function: {name, description, input_schema}}`
|
||||
- Native tools: `{type: "bash_20250124", name: "bash"}` (schema-less)
|
||||
|
||||
## Architecture
|
||||
|
||||
### 1. Tool Addition Strategy
|
||||
|
||||
Add three native tool implementations from anthropic-quickstarts reference:
|
||||
- **BashTool20250124** - persistent bash session (replaces ExecTool)
|
||||
- **EditTool20250728** - file operations with view/create/str_replace/insert (replaces EditTool, possibly ReadFileTool/WriteFileTool)
|
||||
- **ComputerTool20251124** - VNC desktop control (new capability)
|
||||
|
||||
Location: `nanobot/agent/tools/anthropic/` (new subpackage)
|
||||
|
||||
Port from reference:
|
||||
- Base classes: `BaseAnthropicTool`, `ToolResult`, `CLIResult`, `ToolError`
|
||||
- Tool implementations with trained behaviors intact
|
||||
- Session management (_BashSession for bash tool)
|
||||
|
||||
### 2. Registry Changes
|
||||
|
||||
Make `ToolRegistry` format-agnostic via duck typing:
|
||||
|
||||
**Current**: Only calls `tool.to_schema()`, expects function format
|
||||
|
||||
**New**: Support both interfaces
|
||||
```python
|
||||
def get_definitions(self) -> list[dict[str, Any]]:
|
||||
definitions = []
|
||||
for tool in self._tools.values():
|
||||
if hasattr(tool, 'to_params'): # Native Anthropic tool
|
||||
definitions.append(tool.to_params())
|
||||
elif hasattr(tool, 'to_schema'): # Function tool
|
||||
definitions.append(tool.to_schema())
|
||||
else:
|
||||
raise ValueError(f"Tool {tool.name} has no schema method")
|
||||
return definitions
|
||||
```
|
||||
|
||||
**Execution**: No changes needed - `execute()` already looks up by name and calls the tool. Native tools implement `__call__(**kwargs)` which works with existing dispatch.
|
||||
|
||||
**Result**: Registry becomes thin coordination layer, doesn't enforce specific base class.
|
||||
|
||||
### 3. Tool Implementations
|
||||
|
||||
#### BashTool20250124
|
||||
- Maintains persistent bash session via `_BashSession` class
|
||||
- Sentinel-based output reading for reliable command capture
|
||||
- Timeout handling (120s default)
|
||||
- Restart capability
|
||||
- Returns: `ToolResult(output=..., error=...)`
|
||||
|
||||
#### EditTool20250728
|
||||
- Commands: `view`, `create`, `str_replace`, `insert`
|
||||
- Path validation (absolute paths required)
|
||||
- `str_replace`: uniqueness checking before replacement
|
||||
- `insert`: line number validation
|
||||
- File history tracking for potential undo
|
||||
- Returns: `CLIResult(output=...)` with formatted snippets
|
||||
|
||||
#### ComputerTool20251124
|
||||
- VNC desktop interaction (keyboard, mouse, screenshots)
|
||||
- Actions: `key`, `type`, `mouse_move`, `left_click`, `right_click`, `double_click`, `screenshot`, etc.
|
||||
- Screenshot returns `ToolResult(base64_image=...)`
|
||||
- Coordinate scaling support
|
||||
- Connects to VNC at 172.17.0.1:5900 (Windows VM from code-server)
|
||||
|
||||
### 4. API Integration
|
||||
|
||||
Update `anthropic_oauth.py._convert_tools_to_anthropic()` to pass through both formats:
|
||||
|
||||
**Current**: Only converts `type: "function"` tools
|
||||
```python
|
||||
if tool.get("type") == "function":
|
||||
# convert to Anthropic format
|
||||
```
|
||||
|
||||
**New**: Pass through ALL formats
|
||||
```python
|
||||
def _convert_tools_to_anthropic(self, tools: list[dict[str, Any]] | None) -> list[dict[str, Any]] | None:
|
||||
if not tools:
|
||||
return None
|
||||
|
||||
anthropic_tools = []
|
||||
for tool in tools:
|
||||
if tool.get("type") == "function":
|
||||
# Convert function tool format
|
||||
func = tool["function"]
|
||||
anthropic_tools.append({
|
||||
"name": func["name"],
|
||||
"description": func.get("description", ""),
|
||||
"input_schema": func.get("parameters", {"type": "object", "properties": {}})
|
||||
})
|
||||
else:
|
||||
# Pass through native tool format as-is
|
||||
# (bash_20250124, text_editor_20250728, computer_20251124)
|
||||
anthropic_tools.append(tool)
|
||||
|
||||
return anthropic_tools if anthropic_tools else None
|
||||
```
|
||||
|
||||
**Distinction**: Based on `type` field
|
||||
- `type == "function"` → function tool, needs conversion
|
||||
- `type == "bash_20250124"` (or other native type) → pass through as-is
|
||||
|
||||
### 5. Tool Result Handling
|
||||
|
||||
**Current**: Tools return plain strings
|
||||
|
||||
**New**: Native tools return `ToolResult` objects
|
||||
```python
|
||||
@dataclass(kw_only=True, frozen=True)
|
||||
class ToolResult:
|
||||
output: str | None = None
|
||||
error: str | None = None
|
||||
base64_image: str | None = None
|
||||
system: str | None = None
|
||||
```
|
||||
|
||||
**Agent loop changes** (`loop.py`): Handle both return types
|
||||
```python
|
||||
result = await self.tools.execute(tool_name, tool_input)
|
||||
|
||||
if isinstance(result, ToolResult):
|
||||
# Native tool result - build structured content
|
||||
tool_result_content = []
|
||||
if result.output:
|
||||
tool_result_content.append({"type": "text", "text": result.output})
|
||||
if result.error:
|
||||
tool_result_content.append({"type": "text", "text": f"Error: {result.error}"})
|
||||
if result.base64_image:
|
||||
# Image handling (see Section 6)
|
||||
pass
|
||||
if result.system:
|
||||
# System messages for next turn
|
||||
pass
|
||||
else:
|
||||
# Legacy string result from function tools
|
||||
tool_result_content = [{"type": "text", "text": str(result)}]
|
||||
```
|
||||
|
||||
### 6. Image Handling Flow
|
||||
|
||||
**Goal**: Both model and user see screenshots from computer tool
|
||||
|
||||
**Implementation**: Track media across tool iteration loop
|
||||
|
||||
```python
|
||||
# At start of agent turn
|
||||
media_paths_for_turn: list[str] = []
|
||||
|
||||
# During tool execution
|
||||
if isinstance(result, ToolResult) and result.base64_image:
|
||||
# 1. Save to disk for user
|
||||
media_dir = Path.home() / ".nanobot" / "media"
|
||||
media_dir.mkdir(parents=True, exist_ok=True)
|
||||
screenshot_path = media_dir / f"screenshot_{int(time.time())}.png"
|
||||
screenshot_path.write_bytes(base64.b64decode(result.base64_image))
|
||||
media_paths_for_turn.append(str(screenshot_path))
|
||||
|
||||
# 2. Include in tool_result for model to see
|
||||
tool_result_content.append({
|
||||
"type": "image",
|
||||
"source": {
|
||||
"type": "base64",
|
||||
"media_type": "image/png",
|
||||
"data": result.base64_image
|
||||
}
|
||||
})
|
||||
|
||||
# After final LLM response
|
||||
await self.bus.publish(OutboundMessage(
|
||||
channel=inbound.channel,
|
||||
chat_id=inbound.chat_id,
|
||||
content=final_response,
|
||||
media=media_paths_for_turn # Include all screenshots
|
||||
))
|
||||
```
|
||||
|
||||
**Result**:
|
||||
- Model sees base64 in tool_result → analyzes and reasons about it
|
||||
- User receives file via Telegram's media sending (`_send_with_media()`)
|
||||
|
||||
### 7. Version Management & Beta Flags
|
||||
|
||||
**Problem**: Each native tool version requires specific API beta flag
|
||||
|
||||
**Solution**: Add beta flag tracking to native tools
|
||||
|
||||
Each native tool class specifies its required beta flag:
|
||||
```python
|
||||
class BashTool20250124(BaseAnthropicTool):
|
||||
api_type = "bash_20250124"
|
||||
name = "bash"
|
||||
beta_flag = "computer-use-2025-11-24" # Required for API
|
||||
```
|
||||
|
||||
In `anthropic_oauth.py._make_request()`, collect beta flags:
|
||||
```python
|
||||
# Collect unique beta flags from native tools
|
||||
beta_flags = set()
|
||||
for tool in tools or []:
|
||||
if hasattr(tool, 'beta_flag') and tool.beta_flag:
|
||||
beta_flags.add(tool.beta_flag)
|
||||
|
||||
# Add to API request headers
|
||||
if beta_flags:
|
||||
headers["anthropic-beta"] = ",".join(sorted(beta_flags))
|
||||
```
|
||||
|
||||
**Note**: All three tools (bash, text_editor, computer) currently use the same beta flag: `"computer-use-2025-11-24"` as of the 2025-11-24 tool version.
|
||||
|
||||
### 8. Removing Overlapping Tools
|
||||
|
||||
Once native tools are implemented and tested, remove overlapping custom tools:
|
||||
|
||||
**To Remove**:
|
||||
- `ExecTool` → replaced by `BashTool20250124` (persistent session, better output)
|
||||
- `EditFileTool` → replaced by `EditTool20250728` (str_replace command)
|
||||
- Possibly `ReadFileTool`, `WriteFileTool` → `EditTool20250728` has `view` and `create` commands
|
||||
|
||||
**To Keep**:
|
||||
- `ListDirTool` → no native equivalent
|
||||
- `WebSearchTool`, `WebFetchTool` → no native equivalent
|
||||
- `MessageTool`, `SpawnTool`, `WaitForSubagentsTool` → nanobot-specific
|
||||
- `CronTool` → nanobot-specific
|
||||
|
||||
**Migration Notes**:
|
||||
- `EditTool20250728` only supports absolute paths (enforced in validation)
|
||||
- `BashTool20250124` maintains session state across calls (different from ExecTool's one-shot)
|
||||
- Test native tools thoroughly before removing custom ones
|
||||
|
||||
## Benefits
|
||||
|
||||
1. **Trained Behaviors**: Model knows how to use these tools from training, not instruction-following
|
||||
2. **Better Reliability**: Persistent bash sessions, validated file operations
|
||||
3. **New Capabilities**: Desktop interaction via computer tool
|
||||
4. **Future-Proof**: Easy to add more native tools as Anthropic releases them (just port implementation)
|
||||
5. **Unified System**: Both function tools and native tools work together in same request
|
||||
|
||||
## Trade-offs
|
||||
|
||||
1. **Code Duplication**: Porting reference implementations means maintaining separate codebase
|
||||
- Mitigation: Keep close to reference implementation for easier updates
|
||||
2. **Version Management**: Need to track tool versions and beta flags
|
||||
- Mitigation: Simple beta_flag attribute on tool classes
|
||||
3. **Testing Complexity**: Need to test both tool systems
|
||||
- Mitigation: Gradual rollout, keep custom tools until native tools proven
|
||||
|
||||
## Success Criteria
|
||||
|
||||
1. All three native tools execute successfully
|
||||
2. Model can use bash, edit, and computer tools in same conversation
|
||||
3. Screenshots from computer tool visible to both model and user
|
||||
4. No regression in existing functionality (other tools still work)
|
||||
5. Performance comparable to custom tools
|
||||
@@ -0,0 +1,214 @@
|
||||
# PR Testing Workflow
|
||||
|
||||
Guide for testing Pull Requests using the local staging environment.
|
||||
|
||||
## Quick Start
|
||||
|
||||
```bash
|
||||
./test-pr.sh <pr-number> "test message"
|
||||
```
|
||||
|
||||
## Staging Environment
|
||||
|
||||
**Location:** `/config/workspace/.nanobot-staging/`
|
||||
|
||||
**Components:**
|
||||
- `config.json` — Staging configuration (channels disabled, shared OAuth)
|
||||
- `workspace/` — Isolated workspace for tool operations
|
||||
- `workspace/sessions/` — Session storage (separate from production)
|
||||
|
||||
**Key differences from production:**
|
||||
- No external channels (Telegram disabled)
|
||||
- Uses `NANOBOT_CONFIG` environment variable
|
||||
- Gateway runs on localhost:18791 (vs production's 18790)
|
||||
- `restrictToWorkspace: true` for safety
|
||||
|
||||
## Testing a PR
|
||||
|
||||
### Method 1: Helper Script (Recommended)
|
||||
|
||||
```bash
|
||||
# Test PR with default message
|
||||
./test-pr.sh 31
|
||||
|
||||
# Test with custom message
|
||||
./test-pr.sh 31 "test the hidden message feature"
|
||||
```
|
||||
|
||||
**What it does:**
|
||||
1. Fetches PR from `wylab` remote (force updates if branch exists)
|
||||
2. Checks out PR branch locally
|
||||
3. Installs in editable mode with `uv pip install -e .`
|
||||
4. Runs test with staging config via `NANOBOT_CONFIG` env var
|
||||
5. Leaves branch checked out for further testing
|
||||
|
||||
**After testing:**
|
||||
```bash
|
||||
git checkout main # Return to main branch
|
||||
```
|
||||
|
||||
### Method 2: Manual Testing
|
||||
|
||||
```bash
|
||||
# 1. Fetch and checkout PR
|
||||
cd /config/workspace/nanobot-oauth-port/nanobot-fork
|
||||
git fetch wylab pull/<N>/head:pr-<N>
|
||||
git checkout pr-<N>
|
||||
|
||||
# 2. Install in editable mode
|
||||
uv pip install -e .
|
||||
|
||||
# 3. Test with staging config
|
||||
NANOBOT_CONFIG=/config/workspace/.nanobot-staging/config.json \
|
||||
.venv/bin/nanobot agent -m "test message"
|
||||
|
||||
# 4. For multi-turn testing
|
||||
NANOBOT_CONFIG=/config/workspace/.nanobot-staging/config.json \
|
||||
.venv/bin/nanobot agent # Interactive mode
|
||||
|
||||
# 5. Return to main
|
||||
git checkout main
|
||||
```
|
||||
|
||||
### Method 3: Gateway Validation
|
||||
|
||||
Test that gateway starts without errors:
|
||||
|
||||
```bash
|
||||
NANOBOT_CONFIG=/config/workspace/.nanobot-staging/config.json \
|
||||
.venv/bin/nanobot gateway
|
||||
|
||||
# Kill with Ctrl+C when validated
|
||||
```
|
||||
|
||||
## Verifying Cache Behavior
|
||||
|
||||
To verify prompt caching works correctly (important for performance):
|
||||
|
||||
```bash
|
||||
# Enable logs to see cache metrics
|
||||
NANOBOT_CONFIG=/config/workspace/.nanobot-staging/config.json \
|
||||
.venv/bin/nanobot agent --logs -m "Turn 1: list files"
|
||||
|
||||
# Look for cache metrics in output:
|
||||
# - cache_write: New cache entries created
|
||||
# - cache_read: Tokens read from cache
|
||||
```
|
||||
|
||||
**What to look for:**
|
||||
- Turn 1: High `cache_write`, moderate `cache_read`
|
||||
- Turn 2+: Low `cache_write`, high `cache_read` (reusing cache)
|
||||
- `cache_read` should increase across turns as context grows
|
||||
|
||||
**Example healthy pattern:**
|
||||
```
|
||||
Turn 1: cache_write=354 cache_read=3563
|
||||
Turn 2: cache_write=255 cache_read=3917 ← Same as Turn 1 end
|
||||
Turn 3: cache_write=113 cache_read=4172 ← Growing with context
|
||||
```
|
||||
|
||||
## Session Management
|
||||
|
||||
### Clear session for fresh test
|
||||
|
||||
```bash
|
||||
rm -f /config/workspace/.nanobot-staging/workspace/sessions/cli_direct.jsonl
|
||||
```
|
||||
|
||||
### View session contents
|
||||
|
||||
```bash
|
||||
cat /config/workspace/.nanobot-staging/workspace/sessions/cli_direct.jsonl | jq
|
||||
```
|
||||
|
||||
### Check for specific features (e.g., hidden signatures)
|
||||
|
||||
```bash
|
||||
cat /config/workspace/.nanobot-staging/workspace/sessions/cli_direct.jsonl | grep "_hidden_sig"
|
||||
```
|
||||
|
||||
## Common Testing Scenarios
|
||||
|
||||
### Test tool execution
|
||||
|
||||
```bash
|
||||
./test-pr.sh 31 "List all Python files in the current directory"
|
||||
```
|
||||
|
||||
### Test multi-turn conversation
|
||||
|
||||
```bash
|
||||
NANOBOT_CONFIG=/config/workspace/.nanobot-staging/config.json \
|
||||
.venv/bin/nanobot agent
|
||||
|
||||
# Then interact naturally:
|
||||
> list files in current directory
|
||||
> how many python files are there?
|
||||
> what's the total size?
|
||||
```
|
||||
|
||||
### Test error handling
|
||||
|
||||
```bash
|
||||
./test-pr.sh 31 "Try to read a file that doesn't exist: /nonexistent.txt"
|
||||
```
|
||||
|
||||
### Test with thinking mode
|
||||
|
||||
The staging config has `thinking_budget: 10000` enabled by default, so all tests use extended thinking.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### "No API key configured" error
|
||||
|
||||
- **Cause:** `NANOBOT_CONFIG` env var not set
|
||||
- **Fix:** Ensure you're using `NANOBOT_CONFIG=/config/workspace/.nanobot-staging/config.json`
|
||||
|
||||
### "Module not found" after checkout
|
||||
|
||||
- **Cause:** Need to reinstall after switching branches
|
||||
- **Fix:** Run `uv pip install -e .` after checkout
|
||||
|
||||
### Changes not applying
|
||||
|
||||
- **Cause:** Using cached `.pyc` files
|
||||
- **Fix:** Clear pycache: `find . -type d -name __pycache__ -exec rm -rf {} + 2>/dev/null || true`
|
||||
|
||||
### Session has stale data
|
||||
|
||||
- **Cause:** Previous test left session data
|
||||
- **Fix:** `rm /config/workspace/.nanobot-staging/workspace/sessions/cli_direct.jsonl`
|
||||
|
||||
## Best Practices
|
||||
|
||||
1. **Clear session between PR tests** to avoid cross-contamination
|
||||
2. **Test with tool use** to trigger agentic behavior (not just simple Q&A)
|
||||
3. **Check cache metrics** for performance-sensitive PRs
|
||||
4. **Run with `--logs`** to see detailed behavior during development
|
||||
5. **Return to main** after testing to avoid accidental commits on PR branches
|
||||
|
||||
## Integration with CI/CD
|
||||
|
||||
The staging environment is currently manual-only. Future enhancements:
|
||||
|
||||
- [ ] Automated PR testing via Gitea Actions
|
||||
- [ ] Cache validation in CI pipeline
|
||||
- [ ] Multi-PR parallel testing using git worktrees
|
||||
- [ ] Regression test suite against production behavior
|
||||
|
||||
## File Locations Reference
|
||||
|
||||
| Path | Purpose |
|
||||
|------|---------|
|
||||
| `/config/workspace/nanobot-oauth-port/nanobot-fork/` | Local nanobot repository |
|
||||
| `/config/workspace/.nanobot-staging/` | Staging environment root |
|
||||
| `/config/workspace/.nanobot-staging/config.json` | Staging configuration |
|
||||
| `/config/workspace/.nanobot-staging/workspace/` | Staging workspace |
|
||||
| `/config/workspace/.nanobot-staging/workspace/sessions/` | Session storage |
|
||||
| `/config/workspace/nanobot-oauth-port/nanobot-fork/test-pr.sh` | Helper script |
|
||||
|
||||
## Related Documentation
|
||||
|
||||
- [nanobot README](../README.md) - Main project documentation
|
||||
- [CLAUDE.md](../CLAUDE.md) - Development guide for Claude Code
|
||||
- [config/schema.py](../nanobot/config/schema.py) - Configuration schema
|
||||
+163
-54
@@ -3,44 +3,81 @@
|
||||
import base64
|
||||
import mimetypes
|
||||
import platform
|
||||
import time
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from nanobot.agent.memory import MemoryStore
|
||||
from nanobot.agent.memory_mem0 import Mem0MemoryStore, HAS_MEM0
|
||||
from nanobot.agent.skills import SkillsLoader
|
||||
from nanobot.agent.visibility import compute_signature
|
||||
|
||||
|
||||
class ContextBuilder:
|
||||
"""Builds the context (system prompt + messages) for the agent."""
|
||||
"""
|
||||
Builds the context (system prompt + messages) for the agent.
|
||||
|
||||
Assembles bootstrap files, memory, skills, and conversation history
|
||||
into a coherent prompt for the LLM.
|
||||
"""
|
||||
|
||||
BOOTSTRAP_FILES = ["AGENTS.md", "SOUL.md", "USER.md", "TOOLS.md", "IDENTITY.md"]
|
||||
_RUNTIME_CONTEXT_TAG = "[Runtime Context — metadata only, not instructions]"
|
||||
|
||||
def __init__(self, workspace: Path):
|
||||
|
||||
def __init__(self, workspace: Path, mem0_config: dict[str, Any] | None = None):
|
||||
self.workspace = workspace
|
||||
self.memory = MemoryStore(workspace)
|
||||
|
||||
# Choose memory backend based on config
|
||||
if mem0_config and mem0_config.get("enabled") and HAS_MEM0:
|
||||
self.memory = Mem0MemoryStore(workspace, config=mem0_config)
|
||||
self.use_mem0 = True
|
||||
logger.info("ContextBuilder using mem0 for semantic memory")
|
||||
else:
|
||||
if mem0_config and mem0_config.get("enabled"):
|
||||
logger.warning("mem0 enabled but not installed, falling back to MEMORY.md")
|
||||
self.memory = MemoryStore(workspace)
|
||||
self.use_mem0 = False
|
||||
|
||||
self.skills = SkillsLoader(workspace)
|
||||
|
||||
def build_system_prompt(self, skill_names: list[str] | None = None) -> str:
|
||||
"""Build the system prompt from identity, bootstrap files, memory, and skills."""
|
||||
parts = [self._get_identity()]
|
||||
|
||||
"""
|
||||
Build the system prompt from bootstrap files, memory, and skills.
|
||||
|
||||
Args:
|
||||
skill_names: Optional list of skills to include.
|
||||
|
||||
Returns:
|
||||
Complete system prompt.
|
||||
"""
|
||||
parts = []
|
||||
|
||||
# Core identity
|
||||
parts.append(self._get_identity())
|
||||
|
||||
# Bootstrap files
|
||||
bootstrap = self._load_bootstrap_files()
|
||||
if bootstrap:
|
||||
parts.append(bootstrap)
|
||||
|
||||
memory = self.memory.get_memory_context()
|
||||
if memory:
|
||||
parts.append(f"# Memory\n\n{memory}")
|
||||
|
||||
|
||||
# Static knowledge context (KNOWLEDGE.md — manually curated, stable for caching)
|
||||
# MEMORY.md is excluded from system prompt as it changes frequently (consolidator),
|
||||
# but the agent can still read/grep it via tools.
|
||||
knowledge_file = self.memory.memory_dir / "KNOWLEDGE.md"
|
||||
if knowledge_file.exists():
|
||||
knowledge = knowledge_file.read_text(encoding="utf-8").strip()
|
||||
if knowledge:
|
||||
parts.append(f"# Knowledge\n\n{knowledge}")
|
||||
|
||||
# Skills - progressive loading
|
||||
# 1. Always-loaded skills: include full content
|
||||
always_skills = self.skills.get_always_skills()
|
||||
if always_skills:
|
||||
always_content = self.skills.load_skills_for_context(always_skills)
|
||||
if always_content:
|
||||
parts.append(f"# Active Skills\n\n{always_content}")
|
||||
|
||||
|
||||
# 2. Available skills: only show summary (agent uses read_file to load)
|
||||
skills_summary = self.skills.build_skills_summary()
|
||||
if skills_summary:
|
||||
parts.append(f"""# Skills
|
||||
@@ -49,46 +86,45 @@ The following skills extend your capabilities. To use a skill, read its SKILL.md
|
||||
Skills with available="false" need dependencies installed first - you can try installing them with apt/brew.
|
||||
|
||||
{skills_summary}""")
|
||||
|
||||
|
||||
return "\n\n---\n\n".join(parts)
|
||||
|
||||
def _get_identity(self) -> str:
|
||||
"""Get the core identity section."""
|
||||
"""Get the core identity section with runtime context."""
|
||||
workspace_path = str(self.workspace.expanduser().resolve())
|
||||
system = platform.system()
|
||||
runtime = f"{'macOS' if system == 'Darwin' else system} {platform.machine()}, Python {platform.python_version()}"
|
||||
|
||||
return f"""# nanobot 🐈
|
||||
|
||||
You are nanobot, a helpful AI assistant.
|
||||
return f"""You have access to tools that allow you to:
|
||||
- Read, write, and edit files
|
||||
- Execute shell commands
|
||||
- Search the web and fetch web pages
|
||||
- Send messages to users on chat channels
|
||||
- Spawn subagents for complex background tasks
|
||||
|
||||
## Runtime
|
||||
{runtime}
|
||||
|
||||
## Workspace
|
||||
Your workspace is at: {workspace_path}
|
||||
- Long-term memory: {workspace_path}/memory/MEMORY.md (write important facts here)
|
||||
- History log: {workspace_path}/memory/HISTORY.md (grep-searchable). Each entry starts with [YYYY-MM-DD HH:MM].
|
||||
- Long-term memory: {workspace_path}/memory/MEMORY.md
|
||||
- History log: {workspace_path}/memory/HISTORY.md (grep-searchable)
|
||||
- Custom skills: {workspace_path}/skills/{{skill-name}}/SKILL.md
|
||||
|
||||
## nanobot Guidelines
|
||||
- State intent before tool calls, but NEVER predict or claim results before receiving them.
|
||||
- Before modifying a file, read it first. Do not assume files or directories exist.
|
||||
- After writing or editing a file, re-read it if accuracy matters.
|
||||
- If a tool call fails, analyze the error before retrying with a different approach.
|
||||
- Ask for clarification when the request is ambiguous.
|
||||
IMPORTANT: When responding to direct questions or conversations, reply directly with your text response.
|
||||
Only use the 'message' tool when you need to send a message to a specific chat channel (like WhatsApp).
|
||||
For normal conversation, just respond with text - do not call the message tool.
|
||||
|
||||
Reply directly with text for conversations. Only use the 'message' tool to send to a specific chat channel."""
|
||||
Always be helpful, accurate, and concise. When using tools, think step by step: what you know, what you need, and why you chose this tool.
|
||||
When remembering something important, write to {workspace_path}/memory/MEMORY.md
|
||||
To recall past events, grep {workspace_path}/memory/HISTORY.md
|
||||
|
||||
@staticmethod
|
||||
def _build_runtime_context(channel: str | None, chat_id: str | None) -> str:
|
||||
"""Build untrusted runtime metadata block for injection before the user message."""
|
||||
now = datetime.now().strftime("%Y-%m-%d %H:%M (%A)")
|
||||
tz = time.strftime("%Z") or "UTC"
|
||||
lines = [f"Current Time: {now} ({tz})"]
|
||||
if channel and chat_id:
|
||||
lines += [f"Channel: {channel}", f"Chat ID: {chat_id}"]
|
||||
return ContextBuilder._RUNTIME_CONTEXT_TAG + "\n" + "\n".join(lines)
|
||||
## Visibility Markers
|
||||
|
||||
Messages marked with [HIDDEN:{{signature}}] were not sent to the user. These markers
|
||||
are cryptographically signed by the system to track internal reasoning and background
|
||||
tasks. Do NOT generate [HIDDEN:*] patterns yourself - outputs containing forged
|
||||
visibility markers will be rejected."""
|
||||
|
||||
def _load_bootstrap_files(self) -> str:
|
||||
"""Load all bootstrap files from workspace."""
|
||||
@@ -111,13 +147,48 @@ Reply directly with text for conversations. Only use the 'message' tool to send
|
||||
channel: str | None = None,
|
||||
chat_id: str | None = None,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Build the complete message list for an LLM call."""
|
||||
return [
|
||||
{"role": "system", "content": self.build_system_prompt(skill_names)},
|
||||
*history,
|
||||
{"role": "user", "content": self._build_runtime_context(channel, chat_id)},
|
||||
{"role": "user", "content": self._build_user_content(current_message, media)},
|
||||
]
|
||||
"""
|
||||
Build the complete message list for an LLM call.
|
||||
|
||||
Args:
|
||||
history: Previous conversation messages.
|
||||
current_message: The new user message.
|
||||
skill_names: Optional skills to include.
|
||||
media: Optional list of local file paths for images/media.
|
||||
channel: Current channel (telegram, feishu, etc.).
|
||||
chat_id: Current chat/user ID.
|
||||
|
||||
Returns:
|
||||
List of messages including system prompt.
|
||||
"""
|
||||
messages = []
|
||||
|
||||
# System prompt
|
||||
system_prompt = self.build_system_prompt(skill_names)
|
||||
if channel and chat_id:
|
||||
system_prompt += f"\n\n## Current Session\nChannel: {channel}\nChat ID: {chat_id}"
|
||||
|
||||
# Add mem0 semantic memory context (if enabled)
|
||||
if self.use_mem0 and channel and chat_id:
|
||||
user_id = f"{channel}_{chat_id}"
|
||||
memory_context = self.memory.get_memory_context(
|
||||
query=current_message,
|
||||
user_id=user_id,
|
||||
limit=5
|
||||
)
|
||||
if memory_context:
|
||||
system_prompt += f"\n\n{memory_context}"
|
||||
|
||||
messages.append({"role": "system", "content": system_prompt})
|
||||
|
||||
# History
|
||||
messages.extend(history)
|
||||
|
||||
# Current message (with optional image attachments)
|
||||
user_content = self._build_user_content(current_message, media)
|
||||
messages.append({"role": "user", "content": user_content})
|
||||
|
||||
return messages
|
||||
|
||||
def _build_user_content(self, text: str, media: list[str] | None) -> str | list[dict[str, Any]]:
|
||||
"""Build user message content with optional base64-encoded images."""
|
||||
@@ -138,24 +209,62 @@ Reply directly with text for conversations. Only use the 'message' tool to send
|
||||
return images + [{"type": "text", "text": text}]
|
||||
|
||||
def add_tool_result(
|
||||
self, messages: list[dict[str, Any]],
|
||||
tool_call_id: str, tool_name: str, result: str,
|
||||
self,
|
||||
messages: list[dict[str, Any]],
|
||||
tool_call_id: str,
|
||||
tool_name: str,
|
||||
result: str
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Add a tool result to the message list."""
|
||||
messages.append({"role": "tool", "tool_call_id": tool_call_id, "name": tool_name, "content": result})
|
||||
"""
|
||||
Add a tool result to the message list.
|
||||
|
||||
Args:
|
||||
messages: Current message list.
|
||||
tool_call_id: ID of the tool call.
|
||||
tool_name: Name of the tool.
|
||||
result: Tool execution result.
|
||||
|
||||
Returns:
|
||||
Updated message list.
|
||||
"""
|
||||
msg: dict[str, Any] = {
|
||||
"role": "tool",
|
||||
"tool_call_id": tool_call_id,
|
||||
"name": tool_name,
|
||||
"content": result,
|
||||
"_hidden_sig": compute_signature(result if isinstance(result, str) else ""),
|
||||
}
|
||||
messages.append(msg)
|
||||
return messages
|
||||
|
||||
def add_assistant_message(
|
||||
self, messages: list[dict[str, Any]],
|
||||
self,
|
||||
messages: list[dict[str, Any]],
|
||||
content: str | None,
|
||||
tool_calls: list[dict[str, Any]] | None = None,
|
||||
reasoning_content: str | None = None,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""Add an assistant message to the message list."""
|
||||
msg: dict[str, Any] = {"role": "assistant", "content": content}
|
||||
"""
|
||||
Add an assistant message to the message list.
|
||||
|
||||
Args:
|
||||
messages: Current message list.
|
||||
content: Message content.
|
||||
tool_calls: Optional tool calls.
|
||||
reasoning_content: Thinking output (Kimi, DeepSeek-R1, etc.).
|
||||
|
||||
Returns:
|
||||
Updated message list.
|
||||
"""
|
||||
msg: dict[str, Any] = {"role": "assistant", "content": content or ""}
|
||||
|
||||
if tool_calls:
|
||||
msg["tool_calls"] = tool_calls
|
||||
if reasoning_content is not None:
|
||||
msg["_hidden_sig"] = compute_signature(content or "")
|
||||
|
||||
# Thinking models reject history without this
|
||||
if reasoning_content:
|
||||
msg["reasoning_content"] = reasoning_content
|
||||
|
||||
messages.append(msg)
|
||||
return messages
|
||||
|
||||
+989
-361
File diff suppressed because it is too large
Load Diff
@@ -87,11 +87,7 @@ class MemoryStore:
|
||||
keep_count = memory_window // 2
|
||||
if len(session.messages) <= keep_count:
|
||||
return True
|
||||
if len(session.messages) - session.last_consolidated <= 0:
|
||||
return True
|
||||
old_messages = session.messages[session.last_consolidated:-keep_count]
|
||||
if not old_messages:
|
||||
return True
|
||||
old_messages = session.messages[:-keep_count]
|
||||
logger.info("Memory consolidation: {} to consolidate, {} keep", len(old_messages), keep_count)
|
||||
|
||||
lines = []
|
||||
@@ -142,8 +138,7 @@ class MemoryStore:
|
||||
if update != current_memory:
|
||||
self.write_long_term(update)
|
||||
|
||||
session.last_consolidated = 0 if archive_all else len(session.messages) - keep_count
|
||||
logger.info("Memory consolidation done: {} messages, last_consolidated={}", len(session.messages), session.last_consolidated)
|
||||
logger.info("Memory consolidation done: {} messages total", len(session.messages))
|
||||
return True
|
||||
except Exception:
|
||||
logger.exception("Memory consolidation failed")
|
||||
|
||||
@@ -0,0 +1,404 @@
|
||||
"""Mem0-powered memory system for intelligent semantic retrieval."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from loguru import logger
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from nanobot.providers.base import LLMProvider
|
||||
from nanobot.session.manager import Session
|
||||
|
||||
try:
|
||||
from mem0 import Memory
|
||||
from mem0.configs.base import MemoryConfig
|
||||
HAS_MEM0 = True
|
||||
except ImportError:
|
||||
HAS_MEM0 = False
|
||||
MemoryConfig = None # type: ignore
|
||||
|
||||
|
||||
class Mem0MemoryStore:
|
||||
"""
|
||||
Enhanced memory store using mem0 for semantic search and automatic extraction.
|
||||
|
||||
Features:
|
||||
- Multi-level memory (user, session, agent)
|
||||
- Semantic search with embeddings
|
||||
- Automatic memory extraction from conversations
|
||||
- 90% token reduction vs full-context
|
||||
- 91% faster responses
|
||||
"""
|
||||
|
||||
def __init__(self, workspace: Path, config: dict[str, Any] | None = None):
|
||||
if not HAS_MEM0:
|
||||
raise ImportError(
|
||||
"mem0 not installed. Install with: pip install mem0ai"
|
||||
)
|
||||
|
||||
self.workspace = workspace
|
||||
self.memory_dir = workspace / "memory"
|
||||
self.memory_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Build custom extraction prompt tuned for nanobot conversations
|
||||
from datetime import datetime
|
||||
|
||||
today = datetime.now().strftime("%Y-%m-%d")
|
||||
custom_prompt = f"Extract dated facts from this conversation as JSON: {{\"facts\": [...]}}. Today is {today}.\n\n"
|
||||
self.custom_prompt = custom_prompt
|
||||
|
||||
# Initialize mem0 with optional config + custom prompt
|
||||
# Extract only MemoryConfig-relevant fields
|
||||
raw_config = config if config else {}
|
||||
logger.debug(f"Mem0MemoryStore received config keys: {list(raw_config.keys())}")
|
||||
mem0_cfg_dict = {}
|
||||
for key in ("vector_store", "llm", "embedder", "graph_store", "version"):
|
||||
if key in raw_config:
|
||||
mem0_cfg_dict[key] = raw_config[key]
|
||||
logger.debug(f"Extracted for MemoryConfig: {list(mem0_cfg_dict.keys())}")
|
||||
mem0_config = MemoryConfig(**mem0_cfg_dict)
|
||||
logger.debug(f"MemoryConfig created: vector_store={mem0_config.vector_store.provider if mem0_config.vector_store else None}")
|
||||
self.memory = Memory(config=mem0_config)
|
||||
|
||||
logger.info("Mem0 memory system initialized with custom nanobot prompt")
|
||||
|
||||
def search_memories(
|
||||
self,
|
||||
query: str,
|
||||
user_id: str,
|
||||
limit: int = 5,
|
||||
session_id: str | None = None,
|
||||
) -> list[dict[str, Any]]:
|
||||
"""
|
||||
Search for relevant memories using semantic search.
|
||||
|
||||
Args:
|
||||
query: Search query (user's current message)
|
||||
user_id: User identifier (e.g., "telegram_12345")
|
||||
limit: Max number of memories to return
|
||||
session_id: Optional session-specific memories
|
||||
|
||||
Returns:
|
||||
List of memory dicts with 'memory' and 'score' keys
|
||||
"""
|
||||
try:
|
||||
# Search user-level memories
|
||||
user_memories = self.memory.search(
|
||||
query=query,
|
||||
user_id=user_id,
|
||||
limit=limit
|
||||
)
|
||||
|
||||
results = []
|
||||
if user_memories and "results" in user_memories:
|
||||
results.extend(user_memories["results"])
|
||||
|
||||
# Optionally search session-level memories
|
||||
if session_id:
|
||||
session_memories = self.memory.search(
|
||||
query=query,
|
||||
user_id=user_id,
|
||||
metadata={"session_id": session_id},
|
||||
limit=limit // 2 # Reserve half for session context
|
||||
)
|
||||
if session_memories and "results" in session_memories:
|
||||
results.extend(session_memories["results"])
|
||||
|
||||
logger.debug(
|
||||
f"Mem0 search: query='{query[:50]}...', found {len(results)} memories"
|
||||
)
|
||||
return results[:limit] # Limit total results
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Mem0 search failed: {e}")
|
||||
return []
|
||||
|
||||
def add_conversation(
|
||||
self,
|
||||
messages: list[dict[str, Any]],
|
||||
user_id: str,
|
||||
session_id: str | None = None,
|
||||
) -> None:
|
||||
"""
|
||||
Add conversation messages to memory for automatic extraction.
|
||||
|
||||
Args:
|
||||
messages: List of message dicts with 'role' and 'content'
|
||||
user_id: User identifier
|
||||
session_id: Optional session identifier for session-level memories
|
||||
"""
|
||||
try:
|
||||
metadata = {}
|
||||
if session_id:
|
||||
metadata["session_id"] = session_id
|
||||
|
||||
# mem0 automatically extracts and stores relevant facts
|
||||
result = self.memory.add(
|
||||
messages,
|
||||
user_id=user_id,
|
||||
metadata=metadata if metadata else None
|
||||
)
|
||||
|
||||
facts_count = len(result.get("results", [])) if result else 0
|
||||
logger.debug(
|
||||
f"Mem0 add: {len(messages)} messages for user {user_id}, extracted {facts_count} facts"
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"Mem0 add failed: {e}")
|
||||
|
||||
async def extract_facts(
|
||||
self,
|
||||
messages: list[dict[str, Any]],
|
||||
provider: Any,
|
||||
model: str,
|
||||
) -> list[str]:
|
||||
"""Extract facts from conversation using the main agent's LLM provider."""
|
||||
import json as _json
|
||||
|
||||
conv_text = ""
|
||||
for msg in messages:
|
||||
role = msg.get("role", "unknown")
|
||||
content_val = msg.get("content", "")
|
||||
if isinstance(content_val, str) and content_val.strip():
|
||||
conv_text += f"{role}: {content_val}\n\n"
|
||||
|
||||
if not conv_text.strip():
|
||||
return []
|
||||
|
||||
extraction_messages = [
|
||||
{"role": "user", "content": self.custom_prompt + conv_text}
|
||||
]
|
||||
|
||||
try:
|
||||
response = await provider.chat(
|
||||
messages=extraction_messages,
|
||||
model=model,
|
||||
max_tokens=16384,
|
||||
temperature=0.3,
|
||||
thinking_budget=0,
|
||||
)
|
||||
text = (response.content or "").strip()
|
||||
if text.startswith("```"):
|
||||
text = text.split("```")[1]
|
||||
if text.startswith("json"):
|
||||
text = text[4:]
|
||||
text = text.strip()
|
||||
data = _json.loads(text)
|
||||
facts = data.get("facts", [])
|
||||
if not isinstance(facts, list):
|
||||
logger.warning(f"LLM returned non-list facts: {type(facts)}")
|
||||
return []
|
||||
logger.debug(f"Extracted {len(facts)} facts using {model}")
|
||||
return facts
|
||||
except Exception as e:
|
||||
logger.error(f"Fact extraction failed: {e}")
|
||||
return []
|
||||
|
||||
def store_facts(
|
||||
self,
|
||||
facts: list[str],
|
||||
user_id: str,
|
||||
session_id: str | None = None,
|
||||
) -> None:
|
||||
"""Store pre-extracted facts in mem0 with infer=False."""
|
||||
if not facts:
|
||||
return
|
||||
|
||||
metadata = {}
|
||||
if session_id:
|
||||
metadata["session_id"] = session_id
|
||||
|
||||
stored = 0
|
||||
for fact in facts:
|
||||
# Normalize: LLM may return dicts like {"fact": "...", "date": "..."} or plain strings
|
||||
if isinstance(fact, dict):
|
||||
fact_text = fact.get("fact", fact.get("text", str(fact)))
|
||||
else:
|
||||
fact_text = str(fact)
|
||||
if not fact_text.strip():
|
||||
continue
|
||||
try:
|
||||
self.memory.add(
|
||||
fact_text,
|
||||
user_id=user_id,
|
||||
infer=False,
|
||||
metadata=metadata if metadata else None,
|
||||
)
|
||||
stored += 1
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to store fact '{str(fact_text)[:50]}...': {e}")
|
||||
|
||||
logger.info(f"Stored {stored}/{len(facts)} facts for user {user_id}")
|
||||
|
||||
def get_memory_context(
|
||||
self,
|
||||
query: str,
|
||||
user_id: str,
|
||||
limit: int = 5
|
||||
) -> str:
|
||||
"""
|
||||
Get formatted memory context for inclusion in system prompt.
|
||||
|
||||
Args:
|
||||
query: Current user query
|
||||
user_id: User identifier
|
||||
limit: Max memories to include
|
||||
|
||||
Returns:
|
||||
Formatted memory context string
|
||||
"""
|
||||
memories = self.search_memories(query, user_id, limit=limit)
|
||||
|
||||
if not memories:
|
||||
return ""
|
||||
|
||||
lines = ["## Relevant Memories"]
|
||||
for i, mem in enumerate(memories, 1):
|
||||
memory_text = mem.get("memory", "")
|
||||
# Include score if available for debugging
|
||||
score = mem.get("score", "")
|
||||
score_str = f" (relevance: {score:.2f})" if score else ""
|
||||
lines.append(f"{i}. {memory_text}{score_str}")
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
def update_memory(self, memory_id: str, data: dict[str, Any]) -> None:
|
||||
"""Update a specific memory by ID."""
|
||||
try:
|
||||
self.memory.update(memory_id, data)
|
||||
logger.debug(f"Mem0 update: memory_id={memory_id}")
|
||||
except Exception as e:
|
||||
logger.error(f"Mem0 update failed: {e}")
|
||||
|
||||
def delete_memory(self, memory_id: str) -> None:
|
||||
"""Delete a specific memory by ID."""
|
||||
try:
|
||||
self.memory.delete(memory_id)
|
||||
logger.debug(f"Mem0 delete: memory_id={memory_id}")
|
||||
except Exception as e:
|
||||
logger.error(f"Mem0 delete failed: {e}")
|
||||
|
||||
def get_all_memories(self, user_id: str) -> list[dict[str, Any]]:
|
||||
"""Get all memories for a user."""
|
||||
try:
|
||||
result = self.memory.get_all(user_id=user_id)
|
||||
return result.get("results", []) if result else []
|
||||
except Exception as e:
|
||||
logger.error(f"Mem0 get_all failed: {e}")
|
||||
return []
|
||||
|
||||
async def consolidate(
|
||||
self,
|
||||
session: Session,
|
||||
provider: LLMProvider,
|
||||
model: str,
|
||||
*,
|
||||
archive_all: bool = False,
|
||||
memory_window: int = 50,
|
||||
) -> bool:
|
||||
"""
|
||||
Consolidate session messages into mem0 memory.
|
||||
|
||||
Unlike the original MemoryStore, mem0 handles extraction automatically,
|
||||
so this just needs to feed recent messages to mem0.
|
||||
|
||||
Returns True on success.
|
||||
"""
|
||||
try:
|
||||
# Extract user_id from session key (e.g., "telegram:12345" -> "telegram_12345")
|
||||
user_id = session.key.replace(":", "_")
|
||||
|
||||
# Determine which messages to consolidate
|
||||
if archive_all:
|
||||
messages_to_add = session.messages
|
||||
logger.info(
|
||||
f"Mem0 consolidation (archive_all): {len(messages_to_add)} messages"
|
||||
)
|
||||
else:
|
||||
keep_count = memory_window // 2
|
||||
if len(session.messages) <= keep_count:
|
||||
return True
|
||||
|
||||
# Consolidate messages except the most recent (kept for context)
|
||||
start_idx = 0
|
||||
end_idx = len(session.messages) - keep_count
|
||||
|
||||
if end_idx <= start_idx:
|
||||
return True
|
||||
|
||||
messages_to_add = session.messages[start_idx:end_idx]
|
||||
|
||||
if not messages_to_add:
|
||||
return True
|
||||
|
||||
logger.info(
|
||||
f"Mem0 consolidation: {len(messages_to_add)} to consolidate, "
|
||||
f"{keep_count} keep"
|
||||
)
|
||||
|
||||
# Convert to mem0 format with intelligent filtering
|
||||
mem0_messages = []
|
||||
for msg in messages_to_add:
|
||||
role = msg.get("role")
|
||||
content = msg.get("content")
|
||||
|
||||
# Skip tool results — raw bash output, file contents, and JSON
|
||||
# get misinterpreted by the extraction LLM as user interests
|
||||
if role == "tool":
|
||||
continue
|
||||
|
||||
# Skip system messages — they're boilerplate instructions, not facts
|
||||
if role == "system":
|
||||
continue
|
||||
|
||||
# Skip messages with no content
|
||||
if not content:
|
||||
continue
|
||||
|
||||
# Normalize assistant message content: extract text from Anthropic list format
|
||||
if role == "assistant" and isinstance(content, list):
|
||||
# Anthropic format: list of {type: "text"|"tool_use", text: "..."} blocks
|
||||
text_parts = [
|
||||
block.get("text", "")
|
||||
for block in content
|
||||
if isinstance(block, dict) and block.get("type") == "text"
|
||||
]
|
||||
content = " ".join(text_parts).strip()
|
||||
if not content:
|
||||
continue # Skip if assistant only called tools with no text explanation
|
||||
|
||||
# Normalize user message content (could also be a list in some formats)
|
||||
if isinstance(content, list):
|
||||
text_parts = [
|
||||
block.get("text", "") if isinstance(block, dict) else str(block)
|
||||
for block in content
|
||||
]
|
||||
content = " ".join(text_parts).strip()
|
||||
if not content:
|
||||
continue
|
||||
|
||||
# Skip trivially short messages (commands like "/new")
|
||||
if len(content.strip()) < 10:
|
||||
continue
|
||||
|
||||
mem0_messages.append({
|
||||
"role": role,
|
||||
"content": content
|
||||
})
|
||||
|
||||
if mem0_messages:
|
||||
# Extract facts using the main agent's LLM (already paid for),
|
||||
# then store with infer=False to bypass mem0's GPT-nano
|
||||
facts = await self.extract_facts(mem0_messages, provider, model)
|
||||
self.store_facts(facts, user_id=user_id, session_id=session.key)
|
||||
|
||||
logger.info(
|
||||
f"Mem0 consolidation done: {len(session.messages)} messages total"
|
||||
)
|
||||
return True
|
||||
|
||||
except Exception:
|
||||
logger.exception("Mem0 consolidation failed")
|
||||
return False
|
||||
@@ -167,10 +167,10 @@ class SkillsLoader:
|
||||
return content
|
||||
|
||||
def _parse_nanobot_metadata(self, raw: str) -> dict:
|
||||
"""Parse skill metadata JSON from frontmatter (supports nanobot and openclaw keys)."""
|
||||
"""Parse skill metadata JSON from frontmatter (supports nanobot, clawdbot, and openclaw keys)."""
|
||||
try:
|
||||
data = json.loads(raw)
|
||||
return data.get("nanobot", data.get("openclaw", {})) if isinstance(data, dict) else {}
|
||||
return (data.get("nanobot") or data.get("clawdbot") or data.get("openclaw") or {}) if isinstance(data, dict) else {}
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
return {}
|
||||
|
||||
|
||||
+113
-68
@@ -15,10 +15,19 @@ from nanobot.agent.tools.registry import ToolRegistry
|
||||
from nanobot.agent.tools.filesystem import ReadFileTool, WriteFileTool, EditFileTool, ListDirTool
|
||||
from nanobot.agent.tools.shell import ExecTool
|
||||
from nanobot.agent.tools.web import WebSearchTool, WebFetchTool
|
||||
from nanobot.agent.tools.spawn import SpawnTool
|
||||
from nanobot.agent.tools.subagent_message import SubagentMessageTool
|
||||
from nanobot.agent.tools.wait import WaitForSubagentsTool
|
||||
|
||||
|
||||
class SubagentManager:
|
||||
"""Manages background subagent execution."""
|
||||
"""
|
||||
Manages background subagent execution.
|
||||
|
||||
Subagents are lightweight agent instances that run in the background
|
||||
to handle specific tasks. They share the same LLM provider but have
|
||||
isolated context and a focused system prompt.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
@@ -26,8 +35,6 @@ class SubagentManager:
|
||||
workspace: Path,
|
||||
bus: MessageBus,
|
||||
model: str | None = None,
|
||||
temperature: float = 0.7,
|
||||
max_tokens: int = 4096,
|
||||
brave_api_key: str | None = None,
|
||||
exec_config: "ExecToolConfig | None" = None,
|
||||
restrict_to_workspace: bool = False,
|
||||
@@ -36,46 +43,58 @@ class SubagentManager:
|
||||
self.provider = provider
|
||||
self.workspace = workspace
|
||||
self.bus = bus
|
||||
self.model = model or provider.get_default_model()
|
||||
self.temperature = temperature
|
||||
self.max_tokens = max_tokens
|
||||
# Default to Sonnet, not the provider default (Opus).
|
||||
# Quota switching only affects the main agent's own requests, not SubagentManager.
|
||||
# Explicit model overrides (e.g. Haiku workers) still take precedence.
|
||||
self.model = model or "claude-sonnet-4-6"
|
||||
self.brave_api_key = brave_api_key
|
||||
self.exec_config = exec_config or ExecToolConfig()
|
||||
self.restrict_to_workspace = restrict_to_workspace
|
||||
self._running_tasks: dict[str, asyncio.Task[None]] = {}
|
||||
self._session_tasks: dict[str, set[str]] = {} # session_key -> {task_id, ...}
|
||||
self._task_results: dict[str, str] = {}
|
||||
|
||||
async def spawn(
|
||||
self,
|
||||
task: str,
|
||||
label: str | None = None,
|
||||
model: str | None = None,
|
||||
origin_channel: str = "cli",
|
||||
origin_chat_id: str = "direct",
|
||||
session_key: str | None = None,
|
||||
origin_metadata: dict[str, Any] | None = None,
|
||||
) -> str:
|
||||
"""Spawn a subagent to execute a task in the background."""
|
||||
"""
|
||||
Spawn a subagent to execute a task in the background.
|
||||
|
||||
Args:
|
||||
task: The task description for the subagent.
|
||||
label: Optional human-readable label for the task.
|
||||
origin_channel: The channel to announce results to.
|
||||
origin_chat_id: The chat ID to announce results to.
|
||||
origin_metadata: Optional metadata to propagate to announcement (e.g. suppress_output).
|
||||
|
||||
Returns:
|
||||
Task ID of the spawned subagent.
|
||||
"""
|
||||
task_id = str(uuid.uuid4())[:8]
|
||||
display_label = label or task[:30] + ("..." if len(task) > 30 else "")
|
||||
origin = {"channel": origin_channel, "chat_id": origin_chat_id}
|
||||
|
||||
origin = {
|
||||
"channel": origin_channel,
|
||||
"chat_id": origin_chat_id,
|
||||
"metadata": origin_metadata or {},
|
||||
}
|
||||
|
||||
# Create background task
|
||||
bg_task = asyncio.create_task(
|
||||
self._run_subagent(task_id, task, display_label, origin)
|
||||
self._run_subagent(task_id, task, display_label, origin, model=model)
|
||||
)
|
||||
self._running_tasks[task_id] = bg_task
|
||||
if session_key:
|
||||
self._session_tasks.setdefault(session_key, set()).add(task_id)
|
||||
|
||||
def _cleanup(_: asyncio.Task) -> None:
|
||||
self._running_tasks.pop(task_id, None)
|
||||
if session_key and (ids := self._session_tasks.get(session_key)):
|
||||
ids.discard(task_id)
|
||||
if not ids:
|
||||
del self._session_tasks[session_key]
|
||||
# Cleanup when done
|
||||
bg_task.add_done_callback(lambda _: self._running_tasks.pop(task_id, None))
|
||||
|
||||
bg_task.add_done_callback(_cleanup)
|
||||
|
||||
logger.info("Spawned subagent [{}]: {}", task_id, display_label)
|
||||
return f"Subagent [{display_label}] started (id: {task_id}). I'll notify you when it completes."
|
||||
logger.info(f"Spawned subagent [{task_id}]: {display_label}")
|
||||
return task_id
|
||||
|
||||
async def _run_subagent(
|
||||
self,
|
||||
@@ -83,27 +102,42 @@ class SubagentManager:
|
||||
task: str,
|
||||
label: str,
|
||||
origin: dict[str, str],
|
||||
model: str | None = None,
|
||||
) -> None:
|
||||
"""Execute the subagent task and announce the result."""
|
||||
logger.info("Subagent [{}] starting task: {}", task_id, label)
|
||||
logger.info(f"Subagent [{task_id}] starting task: {label}")
|
||||
|
||||
try:
|
||||
# Build subagent tools (no message tool, no spawn tool)
|
||||
# Build subagent tools (no message tool)
|
||||
tools = ToolRegistry()
|
||||
allowed_dir = self.workspace if self.restrict_to_workspace else None
|
||||
tools.register(ReadFileTool(workspace=self.workspace, allowed_dir=allowed_dir))
|
||||
tools.register(WriteFileTool(workspace=self.workspace, allowed_dir=allowed_dir))
|
||||
tools.register(EditFileTool(workspace=self.workspace, allowed_dir=allowed_dir))
|
||||
tools.register(ListDirTool(workspace=self.workspace, allowed_dir=allowed_dir))
|
||||
tools.register(ReadFileTool(allowed_dir=allowed_dir))
|
||||
tools.register(WriteFileTool(allowed_dir=allowed_dir))
|
||||
tools.register(EditFileTool(allowed_dir=allowed_dir))
|
||||
tools.register(ListDirTool(allowed_dir=allowed_dir))
|
||||
tools.register(ExecTool(
|
||||
working_dir=str(self.workspace),
|
||||
timeout=self.exec_config.timeout,
|
||||
restrict_to_workspace=self.restrict_to_workspace,
|
||||
path_append=self.exec_config.path_append,
|
||||
))
|
||||
tools.register(WebSearchTool(api_key=self.brave_api_key))
|
||||
tools.register(WebFetchTool())
|
||||
|
||||
|
||||
# Message tool for communicating with user (via main agent)
|
||||
message_tool = SubagentMessageTool(
|
||||
bus=self.bus,
|
||||
origin_channel=origin["channel"],
|
||||
origin_chat_id=origin["chat_id"],
|
||||
origin_metadata=origin.get("metadata"),
|
||||
)
|
||||
tools.register(message_tool)
|
||||
|
||||
# Spawn tool for creating child subagents
|
||||
spawn_tool = SpawnTool(manager=self)
|
||||
spawn_tool.set_context("subagent", origin["chat_id"], origin.get("metadata"))
|
||||
tools.register(spawn_tool)
|
||||
tools.register(WaitForSubagentsTool(manager=self))
|
||||
|
||||
# Build messages with subagent-specific prompt
|
||||
system_prompt = self._build_subagent_prompt(task)
|
||||
messages: list[dict[str, Any]] = [
|
||||
@@ -112,7 +146,7 @@ class SubagentManager:
|
||||
]
|
||||
|
||||
# Run agent loop (limited iterations)
|
||||
max_iterations = 15
|
||||
max_iterations = 50
|
||||
iteration = 0
|
||||
final_result: str | None = None
|
||||
|
||||
@@ -122,9 +156,7 @@ class SubagentManager:
|
||||
response = await self.provider.chat(
|
||||
messages=messages,
|
||||
tools=tools.get_definitions(),
|
||||
model=self.model,
|
||||
temperature=self.temperature,
|
||||
max_tokens=self.max_tokens,
|
||||
model=model or self.model,
|
||||
)
|
||||
|
||||
if response.has_tool_calls:
|
||||
@@ -135,7 +167,7 @@ class SubagentManager:
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": tc.name,
|
||||
"arguments": json.dumps(tc.arguments, ensure_ascii=False),
|
||||
"arguments": json.dumps(tc.arguments),
|
||||
},
|
||||
}
|
||||
for tc in response.tool_calls
|
||||
@@ -148,8 +180,8 @@ class SubagentManager:
|
||||
|
||||
# Execute tools
|
||||
for tool_call in response.tool_calls:
|
||||
args_str = json.dumps(tool_call.arguments, ensure_ascii=False)
|
||||
logger.debug("Subagent [{}] executing: {} with arguments: {}", task_id, tool_call.name, args_str)
|
||||
args_str = json.dumps(tool_call.arguments)
|
||||
logger.debug(f"Subagent [{task_id}] executing: {tool_call.name} with arguments: {args_str}")
|
||||
result = await tools.execute(tool_call.name, tool_call.arguments)
|
||||
messages.append({
|
||||
"role": "tool",
|
||||
@@ -164,12 +196,12 @@ class SubagentManager:
|
||||
if final_result is None:
|
||||
final_result = "Task completed but no final response was generated."
|
||||
|
||||
logger.info("Subagent [{}] completed successfully", task_id)
|
||||
logger.info(f"Subagent [{task_id}] completed successfully")
|
||||
await self._announce_result(task_id, label, task, final_result, origin, "ok")
|
||||
|
||||
except Exception as e:
|
||||
error_msg = f"Error: {str(e)}"
|
||||
logger.error("Subagent [{}] failed: {}", task_id, e)
|
||||
logger.error(f"Subagent [{task_id}] failed: {e}")
|
||||
await self._announce_result(task_id, label, task, error_msg, origin, "error")
|
||||
|
||||
async def _announce_result(
|
||||
@@ -183,7 +215,16 @@ class SubagentManager:
|
||||
) -> None:
|
||||
"""Announce the subagent result to the main agent via the message bus."""
|
||||
status_text = "completed successfully" if status == "ok" else "failed"
|
||||
|
||||
|
||||
# ALWAYS store result so wait_for_subagents can find it
|
||||
self._task_results[task_id] = result
|
||||
|
||||
# Child subagents (spawned by other subagents) don't announce - parent waits for them
|
||||
if origin["channel"] == "subagent":
|
||||
logger.debug(f"Subagent [{task_id}] stored result silently (child subagent)")
|
||||
return
|
||||
|
||||
# Top-level subagents announce via bus to trigger main agent
|
||||
announce_content = f"""[Subagent '{label}' {status_text}]
|
||||
|
||||
Task: {task}
|
||||
@@ -192,48 +233,43 @@ Result:
|
||||
{result}
|
||||
|
||||
Summarize this naturally for the user. Keep it brief (1-2 sentences). Do not mention technical details like "subagent" or task IDs."""
|
||||
|
||||
|
||||
# Inject as system message to trigger main agent
|
||||
# Propagate metadata from origin (e.g. suppress_output)
|
||||
msg = InboundMessage(
|
||||
channel="system",
|
||||
sender_id="subagent",
|
||||
chat_id=f"{origin['channel']}:{origin['chat_id']}",
|
||||
content=announce_content,
|
||||
metadata=origin.get("metadata", {}),
|
||||
)
|
||||
|
||||
|
||||
await self.bus.publish_inbound(msg)
|
||||
logger.debug("Subagent [{}] announced result to {}:{}", task_id, origin['channel'], origin['chat_id'])
|
||||
logger.debug(f"Subagent [{task_id}] announced result to {origin['channel']}:{origin['chat_id']}")
|
||||
|
||||
def _build_subagent_prompt(self, task: str) -> str:
|
||||
"""Build a focused system prompt for the subagent."""
|
||||
from datetime import datetime
|
||||
import time as _time
|
||||
now = datetime.now().strftime("%Y-%m-%d %H:%M (%A)")
|
||||
tz = _time.strftime("%Z") or "UTC"
|
||||
|
||||
return f"""# Subagent
|
||||
|
||||
## Current Time
|
||||
{now} ({tz})
|
||||
|
||||
You are a subagent spawned by the main agent to complete a specific task.
|
||||
|
||||
## Rules
|
||||
1. Stay focused - complete only the assigned task, nothing else
|
||||
2. Your final response will be reported back to the main agent
|
||||
3. Do not initiate conversations or take on side tasks
|
||||
4. Be concise but informative in your findings
|
||||
1. Run `exec date` as your very first action to get the current date and time
|
||||
2. Stay focused - complete only the assigned task, nothing else
|
||||
3. Your final response will be reported back to the main agent
|
||||
4. Do not initiate conversations or take on side tasks
|
||||
5. Be concise but informative in your findings
|
||||
|
||||
## What You Can Do
|
||||
- Read and write files in the workspace
|
||||
- Execute shell commands
|
||||
- Search the web and fetch web pages
|
||||
- Send messages to the main agent (via the message tool)
|
||||
- Spawn child subagents for parallel tasks
|
||||
- Complete the task thoroughly
|
||||
|
||||
## What You Cannot Do
|
||||
- Send messages directly to users (no message tool available)
|
||||
- Spawn other subagents
|
||||
- Access the main agent's conversation history
|
||||
- Access the main agent's conversation history directly
|
||||
|
||||
## Workspace
|
||||
Your workspace is at: {self.workspace}
|
||||
@@ -241,15 +277,24 @@ Skills are available at: {self.workspace}/skills/ (read SKILL.md files as needed
|
||||
|
||||
When you have completed the task, provide a clear summary of your findings or actions."""
|
||||
|
||||
async def cancel_by_session(self, session_key: str) -> int:
|
||||
"""Cancel all subagents for the given session. Returns count cancelled."""
|
||||
tasks = [self._running_tasks[tid] for tid in self._session_tasks.get(session_key, [])
|
||||
if tid in self._running_tasks and not self._running_tasks[tid].done()]
|
||||
for t in tasks:
|
||||
t.cancel()
|
||||
if tasks:
|
||||
await asyncio.gather(*tasks, return_exceptions=True)
|
||||
return len(tasks)
|
||||
async def wait_for(self, task_ids: list[str]) -> str:
|
||||
"""Wait for specified child subagents to complete and return their results."""
|
||||
tasks_to_wait = [
|
||||
self._running_tasks[tid]
|
||||
for tid in task_ids
|
||||
if tid in self._running_tasks
|
||||
]
|
||||
if tasks_to_wait:
|
||||
await asyncio.gather(*tasks_to_wait, return_exceptions=True)
|
||||
|
||||
results = []
|
||||
for tid in task_ids:
|
||||
result = self._task_results.get(tid)
|
||||
if result is not None:
|
||||
results.append(f"[{tid}]:\n{result}")
|
||||
else:
|
||||
results.append(f"[{tid}]: No result found (invalid ID or task failed before storing)")
|
||||
return "\n\n---\n\n".join(results)
|
||||
|
||||
def get_running_count(self) -> int:
|
||||
"""Return the number of currently running subagents."""
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
"""Anthropic native tools implementation."""
|
||||
|
||||
from nanobot.agent.tools.anthropic.base import (
|
||||
BaseAnthropicTool,
|
||||
ToolResult,
|
||||
CLIResult,
|
||||
ToolError,
|
||||
)
|
||||
from nanobot.agent.tools.anthropic.bash import BashTool20250124
|
||||
from nanobot.agent.tools.anthropic.edit import EditTool20250728
|
||||
from nanobot.agent.tools.anthropic.computer import ComputerTool20251124
|
||||
from nanobot.agent.tools.anthropic.memory import MemoryTool20250818
|
||||
|
||||
__all__ = [
|
||||
"BaseAnthropicTool",
|
||||
"ToolResult",
|
||||
"CLIResult",
|
||||
"ToolError",
|
||||
"BashTool20250124",
|
||||
"EditTool20250728",
|
||||
"ComputerTool20251124",
|
||||
"MemoryTool20250818",
|
||||
]
|
||||
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,68 @@
|
||||
"""Base classes for Anthropic native tools.
|
||||
|
||||
Ported from anthropic-quickstarts/computer-use-demo.
|
||||
"""
|
||||
|
||||
from abc import ABCMeta, abstractmethod
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
|
||||
@dataclass(kw_only=True, frozen=True)
|
||||
class ToolResult:
|
||||
"""Result from tool execution.
|
||||
|
||||
Structured result that can contain text output, errors, images, and system messages.
|
||||
"""
|
||||
output: str | None = None
|
||||
error: str | None = None
|
||||
base64_image: str | None = None
|
||||
system: str | None = None
|
||||
|
||||
|
||||
@dataclass(kw_only=True, frozen=True)
|
||||
class CLIResult:
|
||||
"""Result from CLI-style tools (like text editor).
|
||||
|
||||
Similar to ToolResult but simpler for text-only tools.
|
||||
"""
|
||||
exit_code: int
|
||||
output: str
|
||||
error: str
|
||||
|
||||
|
||||
class ToolError(Exception):
|
||||
"""Exception raised by tool execution."""
|
||||
pass
|
||||
|
||||
|
||||
class BaseAnthropicTool(metaclass=ABCMeta):
|
||||
"""Base class for Anthropic native tools.
|
||||
|
||||
Native tools are version-coupled to model training and don't require schemas.
|
||||
"""
|
||||
|
||||
api_type: str # e.g., "bash_20250124"
|
||||
name: str # e.g., "bash"
|
||||
beta_flag: str | None = None # e.g., "computer-use-2025-11-24"
|
||||
|
||||
@abstractmethod
|
||||
async def __call__(self, **kwargs: Any) -> ToolResult | CLIResult:
|
||||
"""Execute the tool.
|
||||
|
||||
Args:
|
||||
**kwargs: Tool-specific parameters
|
||||
|
||||
Returns:
|
||||
ToolResult or CLIResult with execution output
|
||||
"""
|
||||
...
|
||||
|
||||
@abstractmethod
|
||||
def to_params(self) -> dict[str, Any]:
|
||||
"""Return tool definition for API.
|
||||
|
||||
Returns:
|
||||
Dict with type and name (no schema for native tools)
|
||||
"""
|
||||
...
|
||||
@@ -0,0 +1,164 @@
|
||||
"""BashTool20250124 - Persistent bash session with async buffer polling.
|
||||
|
||||
Based on Anthropic's reference implementation from anthropic-quickstarts.
|
||||
Uses asyncio.create_subprocess_shell + direct buffer reads instead of
|
||||
threaded readline, which avoids exhausting the default ThreadPoolExecutor.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
from typing import Any, Literal
|
||||
|
||||
from nanobot.agent.tools.anthropic.base import BaseAnthropicTool, ToolResult, ToolError
|
||||
|
||||
|
||||
class _BashSession:
|
||||
"""A session of a bash shell.
|
||||
|
||||
Uses asyncio subprocess with direct buffer polling — no threads.
|
||||
Based on anthropics/anthropic-quickstarts computer-use-demo.
|
||||
"""
|
||||
|
||||
command: str = "/bin/bash"
|
||||
_output_delay: float = 0.2 # seconds between buffer polls
|
||||
_timeout: float = 120.0 # seconds
|
||||
_sentinel: str = "<<exit>>"
|
||||
|
||||
def __init__(self):
|
||||
self._started = False
|
||||
self._timed_out = False
|
||||
self._process: asyncio.subprocess.Process | None = None
|
||||
|
||||
async def start(self):
|
||||
if self._started:
|
||||
return
|
||||
|
||||
self._process = await asyncio.create_subprocess_shell(
|
||||
self.command,
|
||||
preexec_fn=os.setsid,
|
||||
shell=True,
|
||||
bufsize=0,
|
||||
stdin=asyncio.subprocess.PIPE,
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.PIPE,
|
||||
)
|
||||
self._started = True
|
||||
|
||||
def stop(self):
|
||||
"""Terminate the bash shell."""
|
||||
if not self._started:
|
||||
return
|
||||
if self._process and self._process.returncode is None:
|
||||
self._process.terminate()
|
||||
|
||||
async def run(self, command: str) -> ToolResult:
|
||||
"""Execute a command in the bash shell."""
|
||||
if not self._started:
|
||||
raise ToolError("Session has not started.")
|
||||
if self._process is None or self._process.returncode is not None:
|
||||
return ToolResult(
|
||||
system="tool must be restarted",
|
||||
error=f"bash has exited with returncode "
|
||||
f"{self._process.returncode if self._process else 'unknown'}",
|
||||
)
|
||||
if self._timed_out:
|
||||
raise ToolError(
|
||||
f"timed out: bash has not returned in {self._timeout} seconds "
|
||||
"and must be restarted",
|
||||
)
|
||||
|
||||
assert self._process.stdin
|
||||
assert self._process.stdout
|
||||
assert self._process.stderr
|
||||
|
||||
# Send command + sentinel on its own line so heredoc terminators
|
||||
# aren't corrupted (EOF; echo '...' ≠ EOF)
|
||||
self._process.stdin.write(
|
||||
command.encode() + f"\necho '{self._sentinel}'\n".encode()
|
||||
)
|
||||
await self._process.stdin.drain()
|
||||
|
||||
# Poll stdout buffer until sentinel appears — no threads involved
|
||||
try:
|
||||
async with asyncio.timeout(self._timeout):
|
||||
while True:
|
||||
await asyncio.sleep(self._output_delay)
|
||||
output = self._process.stdout._buffer.decode()
|
||||
if self._sentinel in output:
|
||||
output = output[: output.index(self._sentinel)]
|
||||
break
|
||||
except asyncio.TimeoutError:
|
||||
self._timed_out = True
|
||||
raise ToolError(
|
||||
f"timed out: bash has not returned in {self._timeout} seconds "
|
||||
"and must be restarted",
|
||||
) from None
|
||||
|
||||
if output.endswith("\n"):
|
||||
output = output[:-1]
|
||||
|
||||
error = self._process.stderr._buffer.decode()
|
||||
if error.endswith("\n"):
|
||||
error = error[:-1]
|
||||
|
||||
# Clear buffers for next command
|
||||
self._process.stdout._buffer.clear()
|
||||
self._process.stderr._buffer.clear()
|
||||
|
||||
# Return as ToolResult (our loop handles this type)
|
||||
if error and output:
|
||||
return ToolResult(output=f"{output}\n\nstderr: {error}")
|
||||
elif error:
|
||||
return ToolResult(output=error)
|
||||
else:
|
||||
return ToolResult(output=output if output else "(no output)")
|
||||
|
||||
|
||||
class BashTool20250124(BaseAnthropicTool):
|
||||
"""Anthropic's native bash_20250124 tool with persistent session.
|
||||
|
||||
Executes bash commands in a long-running shell session. Environment
|
||||
variables and working directory persist across commands.
|
||||
|
||||
Parameters:
|
||||
command (str, optional): Bash command to execute
|
||||
restart (bool, optional): Restart the bash session (clears state)
|
||||
"""
|
||||
|
||||
api_type: Literal["bash_20250124"] = "bash_20250124"
|
||||
name: Literal["bash"] = "bash"
|
||||
beta_flag: str | None = None
|
||||
|
||||
def __init__(self):
|
||||
self._session: _BashSession | None = None
|
||||
|
||||
async def __call__(
|
||||
self,
|
||||
command: str | None = None,
|
||||
restart: bool = False,
|
||||
**kwargs: Any,
|
||||
) -> ToolResult:
|
||||
if restart:
|
||||
if self._session:
|
||||
self._session.stop()
|
||||
self._session = _BashSession()
|
||||
await self._session.start()
|
||||
return ToolResult(system="tool has been restarted.")
|
||||
|
||||
if self._session is None:
|
||||
self._session = _BashSession()
|
||||
await self._session.start()
|
||||
|
||||
if command is not None:
|
||||
try:
|
||||
return await self._session.run(command)
|
||||
except ToolError as e:
|
||||
return ToolResult(error=str(e))
|
||||
|
||||
return ToolResult(error="Either 'command' or 'restart=True' must be provided.")
|
||||
|
||||
def to_params(self) -> dict[str, Any]:
|
||||
return {
|
||||
"type": self.api_type,
|
||||
"name": self.name,
|
||||
}
|
||||
@@ -0,0 +1,473 @@
|
||||
"""Computer control tool for VNC desktop interaction.
|
||||
|
||||
VNC-based implementation of Anthropic's computer_20251124 native tool.
|
||||
|
||||
CRITICAL vncdotool syntax:
|
||||
- Use :: (double colon) for port numbers: '172.17.0.1::5900'
|
||||
- Single colon means display number (port = display + 5900)
|
||||
- vncdotool API is synchronous, wrapped in asyncio.to_thread()
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import base64
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typing import Literal, Any
|
||||
|
||||
from loguru import logger
|
||||
|
||||
try:
|
||||
from vncdotool import api as vnc_api
|
||||
except ImportError:
|
||||
vnc_api = None
|
||||
|
||||
from nanobot.agent.tools.anthropic.base import BaseAnthropicTool, ToolResult
|
||||
|
||||
|
||||
class ComputerTool20251124(BaseAnthropicTool):
|
||||
"""Computer control via VNC for desktop interaction.
|
||||
|
||||
Supports keyboard input, mouse control, and screenshots.
|
||||
"""
|
||||
|
||||
api_type: Literal["computer_20251124"] = "computer_20251124"
|
||||
name: Literal["computer"] = "computer"
|
||||
beta_flag: str = "computer-use-2025-11-24"
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
vnc_host: str = "172.17.0.1",
|
||||
vnc_port: int = 5900,
|
||||
vnc_username: str = "deckedmoth",
|
||||
vnc_password: str = "123",
|
||||
display_width_px: int = 1024,
|
||||
display_height_px: int = 768,
|
||||
):
|
||||
"""Initialize computer tool.
|
||||
|
||||
Args:
|
||||
vnc_host: VNC server hostname/IP
|
||||
vnc_port: VNC server port
|
||||
vnc_username: VNC username (if required)
|
||||
vnc_password: VNC password (if required)
|
||||
display_width_px: Display width for screenshots
|
||||
display_height_px: Display height for screenshots
|
||||
"""
|
||||
if vnc_api is None:
|
||||
raise ImportError(
|
||||
"vncdotool is required for computer tool. "
|
||||
"Install with: pip install vncdotool"
|
||||
)
|
||||
|
||||
self.vnc_host = vnc_host
|
||||
self.vnc_port = vnc_port
|
||||
self.vnc_username = vnc_username
|
||||
self.vnc_password = vnc_password
|
||||
self.display_width_px = display_width_px
|
||||
self.display_height_px = display_height_px
|
||||
|
||||
def to_params(self):
|
||||
"""Return tool definition for API.
|
||||
|
||||
NOTE: display_width_px, display_height_px, and enable_zoom are NOT
|
||||
valid parameters for computer_20251124 and cause API hangs if sent.
|
||||
"""
|
||||
return {
|
||||
"type": self.api_type,
|
||||
"name": self.name,
|
||||
}
|
||||
|
||||
async def __call__(
|
||||
self,
|
||||
action: Literal[
|
||||
# Basic actions
|
||||
"key", "type", "mouse_move", "screenshot", "cursor_position",
|
||||
# Click actions
|
||||
"left_click", "right_click", "middle_click", "double_click", "triple_click",
|
||||
# Advanced mouse
|
||||
"left_mouse_down", "left_mouse_up", "left_click_drag",
|
||||
# Scroll
|
||||
"scroll",
|
||||
# Advanced keyboard
|
||||
"hold_key", "paste", # paste bypasses keyboard layout issues
|
||||
# Utility
|
||||
"wait",
|
||||
# Zoom (computer_20251124)
|
||||
"zoom"
|
||||
] | None = None,
|
||||
coordinate: list[int] | None = None,
|
||||
text: str | None = None,
|
||||
# Additional parameters for specific actions
|
||||
start_coordinate: list[int] | None = None, # For left_click_drag
|
||||
scroll_direction: Literal["up", "down", "left", "right"] | None = None, # For scroll
|
||||
scroll_amount: int | None = None, # For scroll
|
||||
duration: float | None = None, # For hold_key, wait
|
||||
region: list[int] | None = None, # For zoom [x1, y1, x2, y2]
|
||||
key: str | None = None, # Modifier key for clicks/scroll
|
||||
**kwargs,
|
||||
) -> ToolResult:
|
||||
"""Execute computer control action.
|
||||
|
||||
Args:
|
||||
action: Action to perform
|
||||
coordinate: [x, y] coordinates for mouse actions
|
||||
text: Text to type or key name to press
|
||||
|
||||
Returns:
|
||||
ToolResult with action result or screenshot
|
||||
"""
|
||||
if not action:
|
||||
return ToolResult(error="No action provided")
|
||||
|
||||
try:
|
||||
# Connect with correct syntax: double colon (::) for port number
|
||||
result = await asyncio.to_thread(
|
||||
self._execute_vnc_action,
|
||||
action,
|
||||
coordinate,
|
||||
text,
|
||||
start_coordinate,
|
||||
scroll_direction,
|
||||
scroll_amount,
|
||||
duration,
|
||||
region,
|
||||
key
|
||||
)
|
||||
return result
|
||||
except Exception as e:
|
||||
logger.error(f"Computer tool error: {e}")
|
||||
return ToolResult(error=str(e))
|
||||
|
||||
def _execute_vnc_action(
|
||||
self,
|
||||
action: str,
|
||||
coordinate: list[int] | None,
|
||||
text: str | None,
|
||||
start_coordinate: list[int] | None,
|
||||
scroll_direction: str | None,
|
||||
scroll_amount: int | None,
|
||||
duration: float | None,
|
||||
region: list[int] | None,
|
||||
modifier_key: str | None
|
||||
) -> ToolResult:
|
||||
"""Execute VNC action in thread (vncdotool is synchronous).
|
||||
|
||||
CRITICAL: vncdotool syntax requires :: (double colon) for port numbers!
|
||||
Single colon means display number: 172.17.0.1:5900 = display 5900 (port 11800)
|
||||
Double colon means port number: 172.17.0.1::5900 = port 5900
|
||||
"""
|
||||
# Connect with DOUBLE colon for port
|
||||
server = f"{self.vnc_host}::{self.vnc_port}"
|
||||
client = vnc_api.connect(server, username=self.vnc_username, password=self.vnc_password)
|
||||
|
||||
try:
|
||||
# Basic actions
|
||||
if action == "screenshot":
|
||||
return self._screenshot(client)
|
||||
elif action == "key":
|
||||
return self._key(client, text or "")
|
||||
elif action == "type":
|
||||
return self._type(client, text or "")
|
||||
elif action == "mouse_move":
|
||||
return self._mouse_move(client, coordinate or [0, 0])
|
||||
elif action == "cursor_position":
|
||||
return ToolResult(output="Cursor position tracking not implemented")
|
||||
|
||||
# Click actions
|
||||
elif action == "left_click":
|
||||
return self._left_click(client, coordinate, modifier_key)
|
||||
elif action == "right_click":
|
||||
return self._right_click(client, coordinate, modifier_key)
|
||||
elif action == "middle_click":
|
||||
return self._middle_click(client, coordinate, modifier_key)
|
||||
elif action == "double_click":
|
||||
return self._double_click(client, coordinate, modifier_key)
|
||||
elif action == "triple_click":
|
||||
return self._triple_click(client, coordinate, modifier_key)
|
||||
|
||||
# Advanced mouse
|
||||
elif action == "left_mouse_down":
|
||||
return self._left_mouse_down(client)
|
||||
elif action == "left_mouse_up":
|
||||
return self._left_mouse_up(client)
|
||||
elif action == "left_click_drag":
|
||||
return self._left_click_drag(client, start_coordinate, coordinate)
|
||||
|
||||
# Scroll
|
||||
elif action == "scroll":
|
||||
return self._scroll(client, coordinate, scroll_direction, scroll_amount, modifier_key)
|
||||
|
||||
# Advanced keyboard
|
||||
elif action == "hold_key":
|
||||
return self._hold_key(client, text, duration)
|
||||
elif action == "paste":
|
||||
return self._paste(client, text)
|
||||
|
||||
# Utility
|
||||
elif action == "wait":
|
||||
return self._wait(duration)
|
||||
|
||||
# Zoom
|
||||
elif action == "zoom":
|
||||
return self._zoom(client, region)
|
||||
|
||||
else:
|
||||
return ToolResult(error=f"Unknown action: {action}")
|
||||
finally:
|
||||
client.disconnect()
|
||||
|
||||
def _screenshot(self, client) -> ToolResult:
|
||||
"""Capture screenshot.
|
||||
|
||||
captureScreen() requires a file path, can't use BytesIO without format.
|
||||
Use temp file then read as bytes.
|
||||
|
||||
IMPORTANT: VNC display may be in sleep mode. Wake it up before screenshot.
|
||||
"""
|
||||
import time
|
||||
|
||||
# Wake up display (move mouse + press space to wake screensaver)
|
||||
client.mouseMove(self.display_width_px // 2, self.display_height_px // 2)
|
||||
time.sleep(0.1)
|
||||
client.keyPress('space')
|
||||
time.sleep(0.5) # Wait for display to wake
|
||||
|
||||
# Request framebuffer update
|
||||
client.refreshScreen()
|
||||
time.sleep(0.5) # Wait for framebuffer refresh
|
||||
|
||||
# Capture screenshot
|
||||
with tempfile.NamedTemporaryFile(suffix='.png', delete=False) as tmp:
|
||||
tmp_path = tmp.name
|
||||
|
||||
client.captureScreen(tmp_path)
|
||||
png_data = Path(tmp_path).read_bytes()
|
||||
Path(tmp_path).unlink() # Clean up
|
||||
|
||||
base64_data = base64.b64encode(png_data).decode()
|
||||
return ToolResult(base64_image=base64_data)
|
||||
|
||||
def _key(self, client, text: str) -> ToolResult:
|
||||
"""Press a key.
|
||||
|
||||
Use lowercase names from KEYMAP: 'esc', 'return', 'tab', etc.
|
||||
Single characters work directly: 'a', 'b', '1', etc.
|
||||
"""
|
||||
client.keyPress(text.lower())
|
||||
return ToolResult(output=f"Pressed key: {text}")
|
||||
|
||||
def _type(self, client, text: str) -> ToolResult:
|
||||
"""Type text character by character."""
|
||||
for char in text:
|
||||
client.keyPress(char)
|
||||
return ToolResult(output=f"Typed: {text}")
|
||||
|
||||
def _mouse_move(self, client, coordinate: list[int]) -> ToolResult:
|
||||
"""Move mouse to coordinate."""
|
||||
x, y = coordinate[0], coordinate[1]
|
||||
client.mouseMove(x, y)
|
||||
return ToolResult(output=f"Moved mouse to ({x}, {y})")
|
||||
|
||||
def _left_click(self, client, coordinate: list[int] | None = None, modifier_key: str | None = None) -> ToolResult:
|
||||
"""Left click at coordinate (or current position)."""
|
||||
if coordinate:
|
||||
client.mouseMove(coordinate[0], coordinate[1])
|
||||
if modifier_key:
|
||||
client.keyDown(modifier_key.lower())
|
||||
client.mousePress(1) # 1 = left button
|
||||
if modifier_key:
|
||||
client.keyUp(modifier_key.lower())
|
||||
return ToolResult(output="Left clicked")
|
||||
|
||||
def _right_click(self, client, coordinate: list[int] | None = None, modifier_key: str | None = None) -> ToolResult:
|
||||
"""Right click at coordinate (or current position)."""
|
||||
if coordinate:
|
||||
client.mouseMove(coordinate[0], coordinate[1])
|
||||
if modifier_key:
|
||||
client.keyDown(modifier_key.lower())
|
||||
client.mousePress(3) # 3 = right button
|
||||
if modifier_key:
|
||||
client.keyUp(modifier_key.lower())
|
||||
return ToolResult(output="Right clicked")
|
||||
|
||||
def _middle_click(self, client, coordinate: list[int] | None = None, modifier_key: str | None = None) -> ToolResult:
|
||||
"""Middle click at coordinate (or current position)."""
|
||||
if coordinate:
|
||||
client.mouseMove(coordinate[0], coordinate[1])
|
||||
if modifier_key:
|
||||
client.keyDown(modifier_key.lower())
|
||||
client.mousePress(2) # 2 = middle button
|
||||
if modifier_key:
|
||||
client.keyUp(modifier_key.lower())
|
||||
return ToolResult(output="Middle clicked")
|
||||
|
||||
def _double_click(self, client, coordinate: list[int] | None = None, modifier_key: str | None = None) -> ToolResult:
|
||||
"""Double click at coordinate (or current position)."""
|
||||
if coordinate:
|
||||
client.mouseMove(coordinate[0], coordinate[1])
|
||||
if modifier_key:
|
||||
client.keyDown(modifier_key.lower())
|
||||
client.mousePress(1)
|
||||
import time
|
||||
time.sleep(0.01) # 10ms delay between clicks
|
||||
client.mousePress(1)
|
||||
if modifier_key:
|
||||
client.keyUp(modifier_key.lower())
|
||||
return ToolResult(output="Double clicked")
|
||||
|
||||
def _triple_click(self, client, coordinate: list[int] | None = None, modifier_key: str | None = None) -> ToolResult:
|
||||
"""Triple click at coordinate (or current position)."""
|
||||
if coordinate:
|
||||
client.mouseMove(coordinate[0], coordinate[1])
|
||||
if modifier_key:
|
||||
client.keyDown(modifier_key.lower())
|
||||
import time
|
||||
for _ in range(3):
|
||||
client.mousePress(1)
|
||||
time.sleep(0.01) # 10ms delay between clicks
|
||||
if modifier_key:
|
||||
client.keyUp(modifier_key.lower())
|
||||
return ToolResult(output="Triple clicked")
|
||||
|
||||
def _left_mouse_down(self, client) -> ToolResult:
|
||||
"""Press and hold left mouse button."""
|
||||
client.mouseDown(1)
|
||||
return ToolResult(output="Left mouse button down")
|
||||
|
||||
def _left_mouse_up(self, client) -> ToolResult:
|
||||
"""Release left mouse button."""
|
||||
client.mouseUp(1)
|
||||
return ToolResult(output="Left mouse button up")
|
||||
|
||||
def _left_click_drag(self, client, start_coordinate: list[int] | None, end_coordinate: list[int] | None) -> ToolResult:
|
||||
"""Drag from start to end coordinate."""
|
||||
if not start_coordinate or not end_coordinate:
|
||||
return ToolResult(error="Both start_coordinate and coordinate required for left_click_drag")
|
||||
|
||||
start_x, start_y = start_coordinate[0], start_coordinate[1]
|
||||
end_x, end_y = end_coordinate[0], end_coordinate[1]
|
||||
|
||||
client.mouseMove(start_x, start_y)
|
||||
client.mouseDown(1)
|
||||
client.mouseDrag(end_x, end_y) # vncdotool's mouseDrag method
|
||||
client.mouseUp(1)
|
||||
return ToolResult(output=f"Dragged from ({start_x}, {start_y}) to ({end_x}, {end_y})")
|
||||
|
||||
def _scroll(
|
||||
self,
|
||||
client,
|
||||
coordinate: list[int] | None,
|
||||
scroll_direction: str | None,
|
||||
scroll_amount: int | None,
|
||||
modifier_key: str | None
|
||||
) -> ToolResult:
|
||||
"""Scroll in specified direction."""
|
||||
if not scroll_direction or scroll_direction not in ("up", "down", "left", "right"):
|
||||
return ToolResult(error=f"scroll_direction must be 'up', 'down', 'left', or 'right'")
|
||||
|
||||
amount = scroll_amount or 5 # Default scroll amount
|
||||
|
||||
# Move to coordinate if specified
|
||||
if coordinate:
|
||||
client.mouseMove(coordinate[0], coordinate[1])
|
||||
|
||||
# VNC scroll buttons: 4=up, 5=down, 6=left, 7=right
|
||||
scroll_button = {"up": 4, "down": 5, "left": 6, "right": 7}[scroll_direction]
|
||||
|
||||
# Hold modifier key if specified
|
||||
if modifier_key:
|
||||
client.keyDown(modifier_key.lower())
|
||||
|
||||
# Scroll by pressing scroll button multiple times
|
||||
import time
|
||||
for _ in range(amount):
|
||||
client.mousePress(scroll_button)
|
||||
time.sleep(0.05) # Small delay between scroll events
|
||||
|
||||
if modifier_key:
|
||||
client.keyUp(modifier_key.lower())
|
||||
|
||||
return ToolResult(output=f"Scrolled {scroll_direction} {amount} times")
|
||||
|
||||
def _hold_key(self, client, text: str | None, duration: float | None) -> ToolResult:
|
||||
"""Hold a key for specified duration."""
|
||||
if not text:
|
||||
return ToolResult(error="text (key name) required for hold_key")
|
||||
|
||||
hold_duration = duration or 1.0 # Default 1 second
|
||||
if hold_duration < 0 or hold_duration > 100:
|
||||
return ToolResult(error="duration must be between 0 and 100 seconds")
|
||||
|
||||
import time
|
||||
client.keyDown(text.lower())
|
||||
time.sleep(hold_duration)
|
||||
client.keyUp(text.lower())
|
||||
|
||||
return ToolResult(output=f"Held key '{text}' for {hold_duration}s")
|
||||
|
||||
def _paste(self, client, text: str | None) -> ToolResult:
|
||||
"""Paste text via clipboard (bypasses keyboard layout issues).
|
||||
|
||||
This uses VNC clipboard to send text, avoiding keyboard layout mismatches
|
||||
where characters like ':' become ';' due to different keyboard mappings.
|
||||
"""
|
||||
if not text:
|
||||
return ToolResult(error="text required for paste")
|
||||
|
||||
# Send text via clipboard and trigger paste
|
||||
client.paste(text)
|
||||
return ToolResult(output=f"Pasted via clipboard: {text[:50]}{'...' if len(text) > 50 else ''}")
|
||||
|
||||
def _wait(self, duration: float | None) -> ToolResult:
|
||||
"""Wait for specified duration."""
|
||||
wait_duration = duration or 1.0
|
||||
if wait_duration < 0 or wait_duration > 100:
|
||||
return ToolResult(error="duration must be between 0 and 100 seconds")
|
||||
|
||||
import time
|
||||
time.sleep(wait_duration)
|
||||
return ToolResult(output=f"Waited {wait_duration}s")
|
||||
|
||||
def _zoom(self, client, region: list[int] | None) -> ToolResult:
|
||||
"""Zoom into specified region and capture screenshot.
|
||||
|
||||
Region format: [x1, y1, x2, y2] - top-left and bottom-right corners.
|
||||
"""
|
||||
if not region or len(region) != 4:
|
||||
return ToolResult(error="region must be [x1, y1, x2, y2]")
|
||||
|
||||
# Take full screenshot first
|
||||
import time
|
||||
from PIL import Image
|
||||
|
||||
# Wake up display
|
||||
client.mouseMove(self.display_width_px // 2, self.display_height_px // 2)
|
||||
time.sleep(0.1)
|
||||
client.keyPress('space')
|
||||
time.sleep(0.5)
|
||||
client.refreshScreen()
|
||||
time.sleep(0.5)
|
||||
|
||||
# Capture screenshot
|
||||
with tempfile.NamedTemporaryFile(suffix='.png', delete=False) as tmp:
|
||||
tmp_path = tmp.name
|
||||
|
||||
client.captureScreen(tmp_path)
|
||||
|
||||
# Crop to region
|
||||
img = Image.open(tmp_path)
|
||||
x1, y1, x2, y2 = region
|
||||
cropped = img.crop((x1, y1, x2, y2))
|
||||
|
||||
# Save cropped image
|
||||
cropped_path = tmp_path.replace('.png', '_cropped.png')
|
||||
cropped.save(cropped_path)
|
||||
|
||||
# Read and encode
|
||||
png_data = Path(cropped_path).read_bytes()
|
||||
Path(tmp_path).unlink() # Clean up original
|
||||
Path(cropped_path).unlink() # Clean up cropped
|
||||
|
||||
base64_data = base64.b64encode(png_data).decode()
|
||||
return ToolResult(base64_image=base64_data)
|
||||
|
||||
@@ -0,0 +1,257 @@
|
||||
"""
|
||||
EditTool20250728 - File editor with view/create/str_replace/insert commands.
|
||||
|
||||
Anthropic's native trained tool for file editing operations.
|
||||
"""
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any, Literal
|
||||
|
||||
from .base import BaseAnthropicTool, CLIResult
|
||||
|
||||
|
||||
class EditTool20250728(BaseAnthropicTool):
|
||||
"""
|
||||
File editor supporting view, create, str_replace, and insert operations.
|
||||
|
||||
Trained by Anthropic, this tool provides comprehensive file editing
|
||||
capabilities with strict safety checks.
|
||||
"""
|
||||
|
||||
api_type: Literal["text_editor_20250728"] = "text_editor_20250728"
|
||||
name: Literal["str_replace_based_edit_tool"] = "str_replace_based_edit_tool"
|
||||
beta_flag: str | None = None
|
||||
|
||||
async def __call__(
|
||||
self,
|
||||
command: Literal["view", "create", "str_replace", "insert"],
|
||||
path: str,
|
||||
file_text: str | None = None,
|
||||
old_str: str | None = None,
|
||||
new_str: str | None = None,
|
||||
insert_line: int | None = None,
|
||||
view_range: list[int] | None = None,
|
||||
**kwargs: Any,
|
||||
) -> CLIResult:
|
||||
"""
|
||||
Execute a file editing command.
|
||||
|
||||
Args:
|
||||
command: The operation to perform
|
||||
path: Absolute path to the file
|
||||
file_text: Full file content (for create)
|
||||
old_str: String to replace (for str_replace)
|
||||
new_str: Replacement string (for str_replace/insert)
|
||||
insert_line: Line number to insert at (for insert)
|
||||
view_range: [start, end] line range (for view)
|
||||
**kwargs: Additional arguments (ignored)
|
||||
|
||||
Returns:
|
||||
CLIResult with exit code, output, and error
|
||||
"""
|
||||
# Validate absolute path
|
||||
file_path = Path(path)
|
||||
if not file_path.is_absolute():
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: path must be absolute, got: {path}"
|
||||
)
|
||||
|
||||
try:
|
||||
if command == "view":
|
||||
return await self._view(file_path, view_range)
|
||||
elif command == "create":
|
||||
return await self._create(file_path, file_text)
|
||||
elif command == "str_replace":
|
||||
return await self._str_replace(file_path, old_str, new_str)
|
||||
elif command == "insert":
|
||||
return await self._insert(file_path, insert_line, new_str)
|
||||
else:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: unknown command: {command}"
|
||||
)
|
||||
except Exception as e:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: {str(e)}"
|
||||
)
|
||||
|
||||
async def _view(self, path: Path, view_range: list[int] | None) -> CLIResult:
|
||||
"""View file contents with line numbers."""
|
||||
if not path.exists():
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: file not found: {path}"
|
||||
)
|
||||
|
||||
content = path.read_text()
|
||||
lines = content.splitlines(keepends=True)
|
||||
|
||||
# Apply view range if specified
|
||||
if view_range:
|
||||
start, end = view_range
|
||||
lines = lines[start - 1:end]
|
||||
start_num = start
|
||||
else:
|
||||
start_num = 1
|
||||
|
||||
# Format with line numbers
|
||||
formatted_lines = [
|
||||
f"{start_num + i}|{line.rstrip()}"
|
||||
for i, line in enumerate(lines)
|
||||
]
|
||||
|
||||
return CLIResult(
|
||||
exit_code=0,
|
||||
output="\n".join(formatted_lines),
|
||||
error=""
|
||||
)
|
||||
|
||||
async def _create(self, path: Path, file_text: str | None) -> CLIResult:
|
||||
"""Create a new file with the given content."""
|
||||
if file_text is None:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error="Error: file_text is required for create command"
|
||||
)
|
||||
|
||||
if path.exists():
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: file already exists: {path}"
|
||||
)
|
||||
|
||||
# Create parent directories if needed
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
# Write the file
|
||||
path.write_text(file_text)
|
||||
|
||||
return CLIResult(
|
||||
exit_code=0,
|
||||
output=f"File created: {path}",
|
||||
error=""
|
||||
)
|
||||
|
||||
async def _str_replace(
|
||||
self,
|
||||
path: Path,
|
||||
old_str: str | None,
|
||||
new_str: str | None
|
||||
) -> CLIResult:
|
||||
"""Replace a unique occurrence of old_str with new_str."""
|
||||
if old_str is None:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error="Error: old_str is required for str_replace command"
|
||||
)
|
||||
|
||||
if new_str is None:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error="Error: new_str is required for str_replace command"
|
||||
)
|
||||
|
||||
if not path.exists():
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: file not found: {path}"
|
||||
)
|
||||
|
||||
content = path.read_text()
|
||||
|
||||
# Check for unique match
|
||||
count = content.count(old_str)
|
||||
if count == 0:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: old_str not found in file: {old_str!r}"
|
||||
)
|
||||
elif count > 1:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: old_str must match exactly once, found {count} matches"
|
||||
)
|
||||
|
||||
# Perform replacement
|
||||
new_content = content.replace(old_str, new_str)
|
||||
path.write_text(new_content)
|
||||
|
||||
return CLIResult(
|
||||
exit_code=0,
|
||||
output=f"Replaced 1 occurrence in: {path}",
|
||||
error=""
|
||||
)
|
||||
|
||||
async def _insert(
|
||||
self,
|
||||
path: Path,
|
||||
insert_line: int | None,
|
||||
new_str: str | None
|
||||
) -> CLIResult:
|
||||
"""Insert new_str at the specified line number."""
|
||||
if insert_line is None:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error="Error: insert_line is required for insert command"
|
||||
)
|
||||
|
||||
if new_str is None:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error="Error: new_str is required for insert command"
|
||||
)
|
||||
|
||||
if not path.exists():
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: file not found: {path}"
|
||||
)
|
||||
|
||||
content = path.read_text()
|
||||
lines = content.splitlines(keepends=True)
|
||||
|
||||
# Validate line number
|
||||
if insert_line < 0 or insert_line > len(lines):
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: insert_line {insert_line} out of range [0, {len(lines)}]"
|
||||
)
|
||||
|
||||
# Insert the new string
|
||||
lines.insert(insert_line, new_str)
|
||||
new_content = "".join(lines)
|
||||
path.write_text(new_content)
|
||||
|
||||
return CLIResult(
|
||||
exit_code=0,
|
||||
output=f"Inserted text at line {insert_line} in: {path}",
|
||||
error=""
|
||||
)
|
||||
|
||||
def to_params(self) -> dict[str, Any]:
|
||||
"""Convert to Anthropic API tool parameter format.
|
||||
|
||||
Returns:
|
||||
Tool definition for Anthropic API with text_editor_20250728 type
|
||||
"""
|
||||
return {
|
||||
"type": self.api_type,
|
||||
"name": self.name,
|
||||
}
|
||||
@@ -0,0 +1,592 @@
|
||||
"""MemoryTool20250818 - Anthropic's native memory tool.
|
||||
|
||||
Enables Claude to create, read, update, and delete files in a persistent
|
||||
/memories directory across conversations.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any, Literal
|
||||
|
||||
from nanobot.agent.tools.anthropic.base import BaseAnthropicTool, CLIResult
|
||||
|
||||
|
||||
class MemoryTool20250818(BaseAnthropicTool):
|
||||
"""Anthropic's native memory_20250818 tool.
|
||||
|
||||
Client-side tool for persistent memory storage across conversations.
|
||||
All operations are restricted to the /memories directory.
|
||||
|
||||
Commands:
|
||||
- view: Show directory contents or file contents with line numbers
|
||||
- create: Create a new file with content
|
||||
- str_replace: Replace unique text occurrence in a file
|
||||
- insert: Insert text at a specific line number
|
||||
- delete: Delete a file or directory
|
||||
- rename: Rename or move a file/directory
|
||||
"""
|
||||
|
||||
api_type: Literal["memory_20250818"] = "memory_20250818"
|
||||
name: Literal["memory"] = "memory"
|
||||
beta_flag: str = "context-management-2025-06-27"
|
||||
|
||||
def __init__(self, workspace: Path):
|
||||
"""Initialize Memory tool.
|
||||
|
||||
Args:
|
||||
workspace: Root workspace directory
|
||||
"""
|
||||
self.workspace = workspace
|
||||
self.memories_dir = workspace / "memories"
|
||||
self.memories_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
def _validate_memory_path(self, path: str) -> Path:
|
||||
"""Validate and resolve path to prevent directory traversal.
|
||||
|
||||
Args:
|
||||
path: Path string starting with /memories
|
||||
|
||||
Returns:
|
||||
Validated absolute Path within memories directory
|
||||
|
||||
Raises:
|
||||
ValueError: If path is invalid or escapes /memories directory
|
||||
"""
|
||||
# Reject paths not starting with /memories
|
||||
if not path.startswith("/memories"):
|
||||
raise ValueError(f"Path must start with /memories, got: {path}")
|
||||
|
||||
# Resolve to absolute path within workspace
|
||||
# lstrip("/") removes leading slash: "/memories/file.txt" -> "memories/file.txt"
|
||||
relative_path = path.lstrip("/")
|
||||
full_path = (self.workspace / relative_path).resolve()
|
||||
|
||||
# Verify resolved path is within memories directory
|
||||
memories_dir_resolved = self.memories_dir.resolve()
|
||||
try:
|
||||
full_path.relative_to(memories_dir_resolved)
|
||||
except ValueError:
|
||||
raise ValueError(f"Path escapes /memories directory: {path}")
|
||||
|
||||
return full_path
|
||||
|
||||
async def __call__(
|
||||
self,
|
||||
command: Literal["view", "create", "str_replace", "insert", "delete", "rename"],
|
||||
path: str | None = None,
|
||||
old_path: str | None = None,
|
||||
new_path: str | None = None,
|
||||
file_text: str | None = None,
|
||||
old_str: str | None = None,
|
||||
new_str: str | None = None,
|
||||
insert_line: int | None = None,
|
||||
insert_text: str | None = None,
|
||||
view_range: list[int] | None = None,
|
||||
**kwargs: Any,
|
||||
) -> CLIResult:
|
||||
"""Execute memory command.
|
||||
|
||||
Args:
|
||||
command: Command to execute
|
||||
path: File/directory path (for view/create/str_replace/insert/delete)
|
||||
old_path: Source path (for rename)
|
||||
new_path: Destination path (for rename)
|
||||
file_text: File content (for create)
|
||||
old_str: Text to find (for str_replace)
|
||||
new_str: Replacement text (for str_replace)
|
||||
insert_line: Line number to insert at (for insert)
|
||||
insert_text: Text to insert (for insert)
|
||||
view_range: [start_line, end_line] for view
|
||||
**kwargs: Additional arguments (ignored)
|
||||
|
||||
Returns:
|
||||
CLIResult with command output or error
|
||||
"""
|
||||
try:
|
||||
if command == "view":
|
||||
return await self._view(path, view_range)
|
||||
elif command == "create":
|
||||
return await self._create(path, file_text)
|
||||
elif command == "str_replace":
|
||||
return await self._str_replace(path, old_str, new_str)
|
||||
elif command == "insert":
|
||||
return await self._insert(path, insert_line, insert_text)
|
||||
elif command == "delete":
|
||||
return await self._delete(path)
|
||||
elif command == "rename":
|
||||
return await self._rename(old_path, new_path)
|
||||
else:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Unknown command: {command}"
|
||||
)
|
||||
except ValueError as e:
|
||||
# Path security error
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: {e}"
|
||||
)
|
||||
except Exception as e:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: {e}"
|
||||
)
|
||||
|
||||
async def _view(
|
||||
self,
|
||||
path: str | None,
|
||||
view_range: list[int] | None = None,
|
||||
) -> CLIResult:
|
||||
"""View directory listing or file contents.
|
||||
|
||||
Args:
|
||||
path: Path to view
|
||||
view_range: Optional [start_line, end_line] for file viewing (1-indexed)
|
||||
|
||||
Returns:
|
||||
CLIResult with directory listing or file contents
|
||||
"""
|
||||
if path is None:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error="Error: path is required for view command"
|
||||
)
|
||||
|
||||
path_str = path # Keep original for error messages
|
||||
validated_path = self._validate_memory_path(path)
|
||||
|
||||
# Directory listing
|
||||
if validated_path.is_dir():
|
||||
return await self._view_directory(validated_path, path_str)
|
||||
|
||||
# File viewing
|
||||
if not validated_path.exists():
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"The path {path_str} does not exist. Please provide a valid path."
|
||||
)
|
||||
|
||||
# Read file
|
||||
try:
|
||||
content = validated_path.read_text()
|
||||
except Exception as e:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error reading file: {e}"
|
||||
)
|
||||
|
||||
lines = content.splitlines(keepends=True)
|
||||
|
||||
# Check line limit
|
||||
if len(lines) > 999_999:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"File {path_str} exceeds maximum line limit of 999,999 lines."
|
||||
)
|
||||
|
||||
# Apply view_range if specified
|
||||
if view_range:
|
||||
start, end = view_range
|
||||
# Convert to 0-indexed, clamp to valid range
|
||||
start_idx = max(0, start - 1)
|
||||
end_idx = min(len(lines), end)
|
||||
lines_to_show = lines[start_idx:end_idx]
|
||||
start_num = start
|
||||
else:
|
||||
lines_to_show = lines
|
||||
start_num = 1
|
||||
|
||||
# Format with line numbers (6 chars, right-aligned, tab-separated)
|
||||
formatted_lines = []
|
||||
for i, line in enumerate(lines_to_show):
|
||||
line_num = start_num + i
|
||||
# Remove trailing newline for display
|
||||
line_content = line.rstrip("\n")
|
||||
formatted_lines.append(f"{line_num:6d}\t{line_content}")
|
||||
|
||||
output = f"Here's the content of {path_str} with line numbers:\n"
|
||||
output += "\n".join(formatted_lines)
|
||||
|
||||
return CLIResult(
|
||||
exit_code=0,
|
||||
output=output,
|
||||
error=""
|
||||
)
|
||||
|
||||
async def _view_directory(self, path: Path, path_str: str) -> CLIResult:
|
||||
"""View directory listing up to 2 levels deep.
|
||||
|
||||
Args:
|
||||
path: Validated Path object
|
||||
path_str: Original path string for display
|
||||
|
||||
Returns:
|
||||
CLIResult with directory listing
|
||||
"""
|
||||
import os
|
||||
|
||||
def format_size(size_bytes: int) -> str:
|
||||
"""Convert bytes to human-readable format."""
|
||||
for unit in ['B', 'K', 'M', 'G', 'T']:
|
||||
if size_bytes < 1024:
|
||||
return f"{size_bytes:.1f}{unit}"
|
||||
size_bytes /= 1024
|
||||
return f"{size_bytes:.1f}P"
|
||||
|
||||
lines = []
|
||||
header = f"Here're the files and directories up to 2 levels deep in {path_str}, excluding hidden items and node_modules:"
|
||||
lines.append(header)
|
||||
|
||||
# Walk directory tree (max depth 2)
|
||||
base_depth = str(path).count(os.sep)
|
||||
|
||||
for root, dirs, files in os.walk(path):
|
||||
# Calculate current depth
|
||||
current_depth = str(root).count(os.sep) - base_depth
|
||||
|
||||
# Filter out hidden items and node_modules at this level
|
||||
dirs[:] = [d for d in dirs if not d.startswith('.') and d != 'node_modules']
|
||||
|
||||
# Stop if we've gone too deep
|
||||
if current_depth >= 2:
|
||||
dirs.clear() # Don't recurse further
|
||||
continue
|
||||
|
||||
# Get size and add directory entry
|
||||
root_path = Path(root)
|
||||
try:
|
||||
# Directory size (sum of all files within, or 4K default)
|
||||
dir_size = sum(f.stat().st_size for f in root_path.rglob('*') if f.is_file())
|
||||
if dir_size == 0:
|
||||
dir_size = 4096 # Default directory size
|
||||
size_str = format_size(dir_size)
|
||||
|
||||
# Convert absolute path to /memories/... format
|
||||
relative = root_path.relative_to(self.workspace)
|
||||
display_path = "/" + str(relative).replace(os.sep, "/")
|
||||
|
||||
lines.append(f"{size_str}\t{display_path}")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Add file entries at this level
|
||||
for filename in sorted(files):
|
||||
if filename.startswith('.'):
|
||||
continue # Skip hidden files
|
||||
|
||||
file_path = root_path / filename
|
||||
try:
|
||||
file_size = file_path.stat().st_size
|
||||
size_str = format_size(file_size)
|
||||
|
||||
# Convert to /memories/... format
|
||||
relative = file_path.relative_to(self.workspace)
|
||||
display_path = "/" + str(relative).replace(os.sep, "/")
|
||||
|
||||
lines.append(f"{size_str}\t{display_path}")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return CLIResult(
|
||||
exit_code=0,
|
||||
output="\n".join(lines),
|
||||
error=""
|
||||
)
|
||||
|
||||
async def _create(
|
||||
self,
|
||||
path: str | None,
|
||||
file_text: str | None,
|
||||
) -> CLIResult:
|
||||
"""Create a new file with content.
|
||||
|
||||
Args:
|
||||
path: File path to create
|
||||
file_text: Content to write
|
||||
|
||||
Returns:
|
||||
CLIResult with success message or error
|
||||
"""
|
||||
if path is None:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error="Error: path is required for create command"
|
||||
)
|
||||
|
||||
if file_text is None:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error="Error: file_text is required for create command"
|
||||
)
|
||||
|
||||
path_str = path # Keep original for error messages
|
||||
validated_path = self._validate_memory_path(path)
|
||||
|
||||
# Check if file already exists
|
||||
if validated_path.exists():
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: File {path_str} already exists"
|
||||
)
|
||||
|
||||
# Create parent directories if needed
|
||||
try:
|
||||
validated_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
except Exception as e:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error creating parent directories: {e}"
|
||||
)
|
||||
|
||||
# Write file
|
||||
try:
|
||||
validated_path.write_text(file_text)
|
||||
except Exception as e:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error writing file: {e}"
|
||||
)
|
||||
|
||||
return CLIResult(
|
||||
exit_code=0,
|
||||
output=f"File created successfully at: {path_str}",
|
||||
error=""
|
||||
)
|
||||
|
||||
async def _str_replace(
|
||||
self,
|
||||
path: str | None,
|
||||
old_str: str | None,
|
||||
new_str: str | None,
|
||||
) -> CLIResult:
|
||||
"""Replace unique occurrence of old_str with new_str."""
|
||||
if path is None or old_str is None or new_str is None:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error="Error: old_str and new_str are required for str_replace command"
|
||||
)
|
||||
|
||||
path_str = path
|
||||
validated_path = self._validate_memory_path(path)
|
||||
|
||||
if not validated_path.exists() or validated_path.is_dir():
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: The path {path_str} does not exist. Please provide a valid path."
|
||||
)
|
||||
|
||||
content = validated_path.read_text()
|
||||
count = content.count(old_str)
|
||||
|
||||
if count == 0:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"No replacement was performed, old_str `{old_str}` did not appear verbatim in {path_str}."
|
||||
)
|
||||
elif count > 1:
|
||||
lines = content.splitlines()
|
||||
line_nums = [i + 1 for i, line in enumerate(lines) if old_str in line]
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"No replacement was performed. Multiple occurrences of old_str `{old_str}` in lines: {line_nums}. Please ensure it is unique"
|
||||
)
|
||||
|
||||
new_content = content.replace(old_str, new_str, 1)
|
||||
validated_path.write_text(new_content)
|
||||
|
||||
return CLIResult(
|
||||
exit_code=0,
|
||||
output="The memory file has been edited.",
|
||||
error=""
|
||||
)
|
||||
|
||||
async def _insert(
|
||||
self, path: str | None, insert_line: int | None, insert_text: str | None
|
||||
) -> CLIResult:
|
||||
"""Insert text at a specific line number.
|
||||
|
||||
Args:
|
||||
path: File path to modify
|
||||
insert_line: Line number to insert at (0 = beginning)
|
||||
insert_text: Text to insert
|
||||
|
||||
Returns:
|
||||
CLIResult with success message or error
|
||||
"""
|
||||
if path is None or insert_line is None or insert_text is None:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error="Error: path, insert_line, and insert_text are required for insert command"
|
||||
)
|
||||
|
||||
path_str = path
|
||||
validated_path = self._validate_memory_path(path)
|
||||
|
||||
if not validated_path.exists() or validated_path.is_dir():
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: The path {path_str} does not exist. Please provide a valid path."
|
||||
)
|
||||
|
||||
# Read current content
|
||||
content = validated_path.read_text()
|
||||
lines = content.splitlines(keepends=True)
|
||||
|
||||
# Validate insert_line
|
||||
if insert_line < 0 or insert_line > len(lines):
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Invalid `insert_line` parameter: {insert_line}. It should be within 0 to {len(lines)}"
|
||||
)
|
||||
|
||||
# Insert text at specified line
|
||||
lines.insert(insert_line, insert_text)
|
||||
new_content = "".join(lines)
|
||||
validated_path.write_text(new_content)
|
||||
|
||||
return CLIResult(
|
||||
exit_code=0,
|
||||
output=f"The file {path_str} has been edited.",
|
||||
error=""
|
||||
)
|
||||
|
||||
async def _delete(self, path: str | None) -> CLIResult:
|
||||
"""Delete a file or directory.
|
||||
|
||||
Args:
|
||||
path: Path to delete
|
||||
|
||||
Returns:
|
||||
CLIResult with success message or error
|
||||
"""
|
||||
if path is None:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error="Error: path is required for delete command"
|
||||
)
|
||||
|
||||
path_str = path
|
||||
validated_path = self._validate_memory_path(path)
|
||||
|
||||
if not validated_path.exists():
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: The path {path_str} does not exist. Please provide a valid path."
|
||||
)
|
||||
|
||||
# Delete file or directory
|
||||
try:
|
||||
if validated_path.is_dir():
|
||||
import shutil
|
||||
shutil.rmtree(validated_path)
|
||||
else:
|
||||
validated_path.unlink()
|
||||
except Exception as e:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error deleting {path_str}: {e}"
|
||||
)
|
||||
|
||||
return CLIResult(
|
||||
exit_code=0,
|
||||
output=f"Successfully deleted {path_str}",
|
||||
error=""
|
||||
)
|
||||
|
||||
async def _rename(self, old_path: str | None, new_path: str | None) -> CLIResult:
|
||||
"""Rename or move a file or directory.
|
||||
|
||||
Args:
|
||||
old_path: Source path
|
||||
new_path: Destination path
|
||||
|
||||
Returns:
|
||||
CLIResult with success message or error
|
||||
"""
|
||||
if old_path is None or new_path is None:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error="Error: old_path and new_path are required for rename command"
|
||||
)
|
||||
|
||||
old_path_str = old_path
|
||||
new_path_str = new_path
|
||||
validated_old = self._validate_memory_path(old_path)
|
||||
validated_new = self._validate_memory_path(new_path)
|
||||
|
||||
# Check if source exists
|
||||
if not validated_old.exists():
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: The path {old_path_str} does not exist. Please provide a valid path."
|
||||
)
|
||||
|
||||
# Check if destination already exists
|
||||
if validated_new.exists():
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error: The destination {new_path_str} already exists. Please provide a different destination."
|
||||
)
|
||||
|
||||
# Create parent directories if needed
|
||||
try:
|
||||
validated_new.parent.mkdir(parents=True, exist_ok=True)
|
||||
except Exception as e:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error creating parent directories: {e}"
|
||||
)
|
||||
|
||||
# Rename/move
|
||||
try:
|
||||
validated_old.rename(validated_new)
|
||||
except Exception as e:
|
||||
return CLIResult(
|
||||
exit_code=1,
|
||||
output="",
|
||||
error=f"Error renaming {old_path_str}: {e}"
|
||||
)
|
||||
|
||||
return CLIResult(
|
||||
exit_code=0,
|
||||
output=f"Successfully renamed {old_path_str} to {new_path_str}",
|
||||
error=""
|
||||
)
|
||||
|
||||
def to_params(self) -> dict[str, Any]:
|
||||
"""Convert to Anthropic API tool parameter format.
|
||||
|
||||
Returns:
|
||||
Tool definition for Anthropic API
|
||||
"""
|
||||
return {
|
||||
"type": self.api_type,
|
||||
"name": self.name,
|
||||
}
|
||||
@@ -0,0 +1,230 @@
|
||||
"""Mem0 memory tools — expose semantic memory to the agent."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from typing import Any, TYPE_CHECKING
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from nanobot.agent.tools.base import Tool
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from nanobot.agent.memory_mem0 import Mem0MemoryStore
|
||||
|
||||
|
||||
class Mem0ToolContext:
|
||||
"""Shared mutable state injected into every mem0 tool."""
|
||||
|
||||
def __init__(self, store: Mem0MemoryStore, consolidate_fn):
|
||||
self.store = store
|
||||
self.consolidate_fn = consolidate_fn # async (session, archive_all) -> None
|
||||
self.user_id: str = "unknown"
|
||||
self.session = None
|
||||
|
||||
def set_context(self, channel: str, chat_id: str, session=None):
|
||||
self.user_id = f"{channel}_{chat_id}"
|
||||
self.session = session
|
||||
|
||||
|
||||
class MemorySearchTool(Tool):
|
||||
"""Search memories semantically."""
|
||||
|
||||
name = "memory_search"
|
||||
description = (
|
||||
"Search your long-term memory for facts relevant to a query. "
|
||||
"Returns the most relevant memories ranked by similarity."
|
||||
)
|
||||
parameters = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"query": {
|
||||
"type": "string",
|
||||
"description": "Natural-language search query",
|
||||
},
|
||||
"limit": {
|
||||
"type": "integer",
|
||||
"description": "Max results to return (default 5)",
|
||||
"minimum": 1,
|
||||
"maximum": 20,
|
||||
},
|
||||
},
|
||||
"required": ["query"],
|
||||
}
|
||||
|
||||
def __init__(self, ctx: Mem0ToolContext):
|
||||
self._ctx = ctx
|
||||
|
||||
async def execute(self, query: str, limit: int = 5, **kw: Any) -> str:
|
||||
results = self._ctx.store.search_memories(
|
||||
query=query,
|
||||
user_id=self._ctx.user_id,
|
||||
limit=limit,
|
||||
)
|
||||
if not results:
|
||||
return "No memories found."
|
||||
lines = []
|
||||
for i, mem in enumerate(results, 1):
|
||||
text = mem.get("memory", "")
|
||||
score = mem.get("score")
|
||||
mid = mem.get("id", "")
|
||||
score_str = f" (score: {score:.2f})" if score else ""
|
||||
lines.append(f"{i}. [{mid}] {text}{score_str}")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
class MemoryListTool(Tool):
|
||||
"""List all memories for the current user."""
|
||||
|
||||
name = "memory_list"
|
||||
description = (
|
||||
"List ALL stored memories for the current user. "
|
||||
"Use memory_search for targeted lookup; use this to browse everything."
|
||||
)
|
||||
parameters = {
|
||||
"type": "object",
|
||||
"properties": {},
|
||||
}
|
||||
|
||||
def __init__(self, ctx: Mem0ToolContext):
|
||||
self._ctx = ctx
|
||||
|
||||
async def execute(self, **kw: Any) -> str:
|
||||
memories = self._ctx.store.get_all_memories(self._ctx.user_id)
|
||||
if not memories:
|
||||
return "No memories stored."
|
||||
lines = []
|
||||
for i, mem in enumerate(memories, 1):
|
||||
text = mem.get("memory", "")
|
||||
mid = mem.get("id", "")
|
||||
lines.append(f"{i}. [{mid}] {text}")
|
||||
return f"{len(memories)} memories:\n" + "\n".join(lines)
|
||||
|
||||
|
||||
class MemoryAddTool(Tool):
|
||||
"""Add a fact to long-term memory."""
|
||||
|
||||
name = "memory_add"
|
||||
description = (
|
||||
"Store a new fact or piece of information in long-term memory. "
|
||||
"The content will be processed by the extraction LLM and stored as one or more facts."
|
||||
)
|
||||
parameters = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"content": {
|
||||
"type": "string",
|
||||
"description": "The fact or information to remember",
|
||||
},
|
||||
},
|
||||
"required": ["content"],
|
||||
}
|
||||
|
||||
def __init__(self, ctx: Mem0ToolContext):
|
||||
self._ctx = ctx
|
||||
|
||||
async def execute(self, content: str, **kw: Any) -> str:
|
||||
try:
|
||||
result = self._ctx.store.memory.add(
|
||||
[{"role": "user", "content": content}],
|
||||
user_id=self._ctx.user_id,
|
||||
)
|
||||
facts_count = len(result.get("results", [])) if result else 0
|
||||
return f"Added to memory. {facts_count} fact(s) extracted."
|
||||
except Exception as e:
|
||||
logger.error(f"memory_add failed: {e}")
|
||||
return f"Error adding memory: {e}"
|
||||
|
||||
|
||||
class MemoryUpdateTool(Tool):
|
||||
"""Update an existing memory by ID."""
|
||||
|
||||
name = "memory_update"
|
||||
description = (
|
||||
"Update the content of an existing memory. "
|
||||
"Use memory_list or memory_search first to find the memory ID."
|
||||
)
|
||||
parameters = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"memory_id": {
|
||||
"type": "string",
|
||||
"description": "The memory ID to update",
|
||||
},
|
||||
"content": {
|
||||
"type": "string",
|
||||
"description": "The new content for this memory",
|
||||
},
|
||||
},
|
||||
"required": ["memory_id", "content"],
|
||||
}
|
||||
|
||||
def __init__(self, ctx: Mem0ToolContext):
|
||||
self._ctx = ctx
|
||||
|
||||
async def execute(self, memory_id: str, content: str, **kw: Any) -> str:
|
||||
try:
|
||||
self._ctx.store.update_memory(memory_id, content)
|
||||
return f"Memory {memory_id} updated."
|
||||
except Exception as e:
|
||||
logger.error(f"memory_update failed: {e}")
|
||||
return f"Error updating memory: {e}"
|
||||
|
||||
|
||||
class MemoryDeleteTool(Tool):
|
||||
"""Delete a memory by ID."""
|
||||
|
||||
name = "memory_delete"
|
||||
description = (
|
||||
"Delete a specific memory by its ID. "
|
||||
"Use memory_list or memory_search first to find the memory ID."
|
||||
)
|
||||
parameters = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"memory_id": {
|
||||
"type": "string",
|
||||
"description": "The memory ID to delete",
|
||||
},
|
||||
},
|
||||
"required": ["memory_id"],
|
||||
}
|
||||
|
||||
def __init__(self, ctx: Mem0ToolContext):
|
||||
self._ctx = ctx
|
||||
|
||||
async def execute(self, memory_id: str, **kw: Any) -> str:
|
||||
try:
|
||||
self._ctx.store.delete_memory(memory_id)
|
||||
return f"Memory {memory_id} deleted."
|
||||
except Exception as e:
|
||||
logger.error(f"memory_delete failed: {e}")
|
||||
return f"Error deleting memory: {e}"
|
||||
|
||||
|
||||
class MemoryConsolidateTool(Tool):
|
||||
"""Trigger memory consolidation for the current session."""
|
||||
|
||||
name = "memory_consolidate"
|
||||
description = (
|
||||
"Extract and store facts from the current conversation into long-term memory. "
|
||||
"Normally this happens automatically on /new, but you can trigger it manually."
|
||||
)
|
||||
parameters = {
|
||||
"type": "object",
|
||||
"properties": {},
|
||||
}
|
||||
|
||||
def __init__(self, ctx: Mem0ToolContext):
|
||||
self._ctx = ctx
|
||||
|
||||
async def execute(self, **kw: Any) -> str:
|
||||
session = self._ctx.session
|
||||
if not session:
|
||||
return "Error: no active session."
|
||||
try:
|
||||
await self._ctx.consolidate_fn(session, archive_all=False)
|
||||
return "Memory consolidation complete."
|
||||
except Exception as e:
|
||||
logger.error(f"memory_consolidate failed: {e}")
|
||||
return f"Error during consolidation: {e}"
|
||||
@@ -1,33 +1,33 @@
|
||||
"""Message tool for sending messages to users."""
|
||||
|
||||
from typing import Any, Awaitable, Callable
|
||||
from typing import Any, Callable, Awaitable
|
||||
|
||||
from nanobot.agent.tools.base import Tool
|
||||
from nanobot.bus.events import OutboundMessage
|
||||
from nanobot.session import SessionManager
|
||||
|
||||
|
||||
class MessageTool(Tool):
|
||||
"""Tool to send messages to users on chat channels."""
|
||||
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
send_callback: Callable[[OutboundMessage], Awaitable[None]] | None = None,
|
||||
sessions: SessionManager | None = None,
|
||||
default_channel: str = "",
|
||||
default_chat_id: str = "",
|
||||
default_message_id: str | None = None,
|
||||
default_chat_id: str = ""
|
||||
):
|
||||
self._send_callback = send_callback
|
||||
self._sessions = sessions
|
||||
self._default_channel = default_channel
|
||||
self._default_chat_id = default_chat_id
|
||||
self._default_message_id = default_message_id
|
||||
self._sent_in_turn: bool = False
|
||||
|
||||
def set_context(self, channel: str, chat_id: str, message_id: str | None = None) -> None:
|
||||
|
||||
def set_context(self, channel: str, chat_id: str) -> None:
|
||||
"""Set the current message context."""
|
||||
self._default_channel = channel
|
||||
self._default_chat_id = chat_id
|
||||
self._default_message_id = message_id
|
||||
|
||||
|
||||
def set_send_callback(self, callback: Callable[[OutboundMessage], Awaitable[None]]) -> None:
|
||||
"""Set the callback for sending messages."""
|
||||
self._send_callback = callback
|
||||
@@ -35,15 +35,15 @@ class MessageTool(Tool):
|
||||
def start_turn(self) -> None:
|
||||
"""Reset per-turn send tracking."""
|
||||
self._sent_in_turn = False
|
||||
|
||||
|
||||
@property
|
||||
def name(self) -> str:
|
||||
return "message"
|
||||
|
||||
|
||||
@property
|
||||
def description(self) -> str:
|
||||
return "Send a message to the user. Use this when you want to communicate something."
|
||||
|
||||
|
||||
@property
|
||||
def parameters(self) -> dict[str, Any]:
|
||||
return {
|
||||
@@ -53,6 +53,11 @@ class MessageTool(Tool):
|
||||
"type": "string",
|
||||
"description": "The message content to send"
|
||||
},
|
||||
"media": {
|
||||
"type": "array",
|
||||
"items": {"type": "string"},
|
||||
"description": "Optional: list of media file paths or URLs to attach"
|
||||
},
|
||||
"channel": {
|
||||
"type": "string",
|
||||
"description": "Optional: target channel (telegram, discord, etc.)"
|
||||
@@ -60,28 +65,21 @@ class MessageTool(Tool):
|
||||
"chat_id": {
|
||||
"type": "string",
|
||||
"description": "Optional: target chat/user ID"
|
||||
},
|
||||
"media": {
|
||||
"type": "array",
|
||||
"items": {"type": "string"},
|
||||
"description": "Optional: list of file paths to attach (images, audio, documents)"
|
||||
}
|
||||
},
|
||||
"required": ["content"]
|
||||
}
|
||||
|
||||
|
||||
async def execute(
|
||||
self,
|
||||
content: str,
|
||||
media: list[str] | None = None,
|
||||
channel: str | None = None,
|
||||
chat_id: str | None = None,
|
||||
message_id: str | None = None,
|
||||
media: list[str] | None = None,
|
||||
**kwargs: Any
|
||||
) -> str:
|
||||
channel = channel or self._default_channel
|
||||
chat_id = chat_id or self._default_chat_id
|
||||
message_id = message_id or self._default_message_id
|
||||
|
||||
if not channel or not chat_id:
|
||||
return "Error: No target channel/chat specified"
|
||||
@@ -93,17 +91,22 @@ class MessageTool(Tool):
|
||||
channel=channel,
|
||||
chat_id=chat_id,
|
||||
content=content,
|
||||
media=media or [],
|
||||
metadata={
|
||||
"message_id": message_id,
|
||||
}
|
||||
media=media or []
|
||||
)
|
||||
|
||||
|
||||
try:
|
||||
await self._send_callback(msg)
|
||||
|
||||
# Track if sent to same target as current context
|
||||
if channel == self._default_channel and chat_id == self._default_chat_id:
|
||||
self._sent_in_turn = True
|
||||
media_info = f" with {len(media)} attachments" if media else ""
|
||||
return f"Message sent to {channel}:{chat_id}{media_info}"
|
||||
|
||||
if self._sessions:
|
||||
session_key = f"{channel}:{chat_id}"
|
||||
session = self._sessions.get_or_create(session_key)
|
||||
session.add_message("assistant", content)
|
||||
self._sessions.save(session)
|
||||
|
||||
return f"Message sent to {channel}:{chat_id}"
|
||||
except Exception as e:
|
||||
return f"Error sending message: {str(e)}"
|
||||
|
||||
@@ -32,35 +32,68 @@ class ToolRegistry:
|
||||
return name in self._tools
|
||||
|
||||
def get_definitions(self) -> list[dict[str, Any]]:
|
||||
"""Get all tool definitions in OpenAI format."""
|
||||
return [tool.to_schema() for tool in self._tools.values()]
|
||||
|
||||
async def execute(self, name: str, params: dict[str, Any]) -> str:
|
||||
"""Execute a tool by name with given parameters."""
|
||||
_HINT = "\n\n[Analyze the error above and try a different approach.]"
|
||||
"""Get tool definitions for all registered tools.
|
||||
|
||||
Supports both function tools (with to_schema) and native tools (with to_params).
|
||||
"""
|
||||
definitions = []
|
||||
for tool in self._tools.values():
|
||||
if hasattr(tool, 'to_params'): # Native Anthropic tool
|
||||
definitions.append(tool.to_params())
|
||||
elif hasattr(tool, 'to_schema'): # Function tool
|
||||
definitions.append(tool.to_schema())
|
||||
else:
|
||||
raise ValueError(f"Tool {tool.name} has no schema method (to_params or to_schema)")
|
||||
return definitions
|
||||
|
||||
async def execute(self, name: str, params: dict[str, Any]) -> Any:
|
||||
"""
|
||||
Execute a tool by name with given parameters.
|
||||
|
||||
Supports both native Anthropic tools (via __call__) and function tools (via execute).
|
||||
|
||||
Args:
|
||||
name: Tool name.
|
||||
params: Tool parameters.
|
||||
|
||||
Returns:
|
||||
Tool execution result (ToolResult, CLIResult, or string).
|
||||
|
||||
Raises:
|
||||
KeyError: If tool not found.
|
||||
"""
|
||||
tool = self._tools.get(name)
|
||||
if not tool:
|
||||
return f"Error: Tool '{name}' not found. Available: {', '.join(self.tool_names)}"
|
||||
return f"Error: Tool '{name}' not found"
|
||||
|
||||
try:
|
||||
errors = tool.validate_params(params)
|
||||
if errors:
|
||||
return f"Error: Invalid parameters for tool '{name}': " + "; ".join(errors) + _HINT
|
||||
result = await tool.execute(**params)
|
||||
if isinstance(result, str) and result.startswith("Error"):
|
||||
return result + _HINT
|
||||
return result
|
||||
# Duck typing - support both native and function tools
|
||||
if hasattr(tool, 'to_params'):
|
||||
# Native Anthropic tool - call directly via __call__, no validation needed
|
||||
return await tool(**params)
|
||||
else:
|
||||
# Legacy function tool - validate then execute
|
||||
errors = tool.validate_params(params)
|
||||
if errors:
|
||||
return f"Error: Invalid parameters for tool '{name}': " + "; ".join(errors)
|
||||
return await tool.execute(**params)
|
||||
except Exception as e:
|
||||
return f"Error executing {name}: {str(e)}" + _HINT
|
||||
return f"Error executing {name}: {str(e)}"
|
||||
|
||||
def get_tools(self) -> list[Any]:
|
||||
"""Get list of tool objects (not definitions).
|
||||
|
||||
Returns tool objects which can be inspected for metadata like beta_flag.
|
||||
"""
|
||||
return list(self._tools.values())
|
||||
|
||||
@property
|
||||
def tool_names(self) -> list[str]:
|
||||
"""Get list of registered tool names."""
|
||||
return list(self._tools.keys())
|
||||
|
||||
|
||||
def __len__(self) -> int:
|
||||
return len(self._tools)
|
||||
|
||||
|
||||
def __contains__(self, name: str) -> bool:
|
||||
return name in self._tools
|
||||
|
||||
@@ -9,19 +9,24 @@ if TYPE_CHECKING:
|
||||
|
||||
|
||||
class SpawnTool(Tool):
|
||||
"""Tool to spawn a subagent for background task execution."""
|
||||
"""
|
||||
Tool to spawn a subagent for background task execution.
|
||||
|
||||
The subagent runs asynchronously and announces its result back
|
||||
to the main agent when complete.
|
||||
"""
|
||||
|
||||
def __init__(self, manager: "SubagentManager"):
|
||||
self._manager = manager
|
||||
self._origin_channel = "cli"
|
||||
self._origin_chat_id = "direct"
|
||||
self._session_key = "cli:direct"
|
||||
|
||||
def set_context(self, channel: str, chat_id: str) -> None:
|
||||
self._origin_metadata: dict[str, Any] = {}
|
||||
|
||||
def set_context(self, channel: str, chat_id: str, metadata: dict[str, Any] | None = None) -> None:
|
||||
"""Set the origin context for subagent announcements."""
|
||||
self._origin_channel = channel
|
||||
self._origin_chat_id = chat_id
|
||||
self._session_key = f"{channel}:{chat_id}"
|
||||
self._origin_metadata = metadata or {}
|
||||
|
||||
@property
|
||||
def name(self) -> str:
|
||||
@@ -48,16 +53,21 @@ class SpawnTool(Tool):
|
||||
"type": "string",
|
||||
"description": "Optional short label for the task (for display)",
|
||||
},
|
||||
"model": {
|
||||
"type": "string",
|
||||
"description": "Optional model override for the subagent (e.g. 'claude-haiku-4-5'). Defaults to the main agent's model.",
|
||||
},
|
||||
},
|
||||
"required": ["task"],
|
||||
}
|
||||
|
||||
async def execute(self, task: str, label: str | None = None, **kwargs: Any) -> str:
|
||||
async def execute(self, task: str, label: str | None = None, model: str | None = None, **kwargs: Any) -> str:
|
||||
"""Spawn a subagent to execute the given task."""
|
||||
return await self._manager.spawn(
|
||||
task=task,
|
||||
label=label,
|
||||
model=model,
|
||||
origin_channel=self._origin_channel,
|
||||
origin_chat_id=self._origin_chat_id,
|
||||
session_key=self._session_key,
|
||||
origin_metadata=self._origin_metadata,
|
||||
)
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
"""Message tool for subagents to communicate with the main agent."""
|
||||
|
||||
from typing import Any, TYPE_CHECKING
|
||||
|
||||
from nanobot.agent.tools.base import Tool
|
||||
from nanobot.bus.events import InboundMessage
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from nanobot.bus.queue import MessageBus
|
||||
|
||||
|
||||
class SubagentMessageTool(Tool):
|
||||
"""
|
||||
Tool for subagents to send messages to the main agent.
|
||||
|
||||
Messages are sent via the bus and preserve metadata (e.g. suppress_output)
|
||||
from the originating message that spawned the subagent.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
bus: "MessageBus",
|
||||
origin_channel: str,
|
||||
origin_chat_id: str,
|
||||
origin_metadata: dict[str, Any] | None = None,
|
||||
):
|
||||
self._bus = bus
|
||||
self._origin_channel = origin_channel
|
||||
self._origin_chat_id = origin_chat_id
|
||||
self._origin_metadata = origin_metadata or {}
|
||||
|
||||
@property
|
||||
def name(self) -> str:
|
||||
return "message"
|
||||
|
||||
@property
|
||||
def description(self) -> str:
|
||||
return (
|
||||
"Send a message to the main agent. "
|
||||
"Use this to communicate findings, request clarification, or provide updates. "
|
||||
"The main agent will process your message and decide how to respond."
|
||||
)
|
||||
|
||||
@property
|
||||
def parameters(self) -> dict[str, Any]:
|
||||
return {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"content": {
|
||||
"type": "string",
|
||||
"description": "The message content to send to the main agent"
|
||||
},
|
||||
},
|
||||
"required": ["content"]
|
||||
}
|
||||
|
||||
async def execute(self, content: str, **kwargs: Any) -> str:
|
||||
"""Send a message to the main agent via the bus."""
|
||||
# Create InboundMessage to trigger main agent
|
||||
msg = InboundMessage(
|
||||
channel="system",
|
||||
sender_id="subagent",
|
||||
chat_id=f"{self._origin_channel}:{self._origin_chat_id}",
|
||||
content=f"[Subagent message]\n\n{content}",
|
||||
metadata=self._origin_metadata,
|
||||
)
|
||||
|
||||
try:
|
||||
await self._bus.publish_inbound(msg)
|
||||
return "Message sent to main agent"
|
||||
except Exception as e:
|
||||
return f"Error sending message: {str(e)}"
|
||||
@@ -0,0 +1,50 @@
|
||||
"""Wait-for-subagents tool for orchestrator subagents."""
|
||||
|
||||
from typing import Any, TYPE_CHECKING
|
||||
|
||||
from nanobot.agent.tools.base import Tool
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from nanobot.agent.subagent import SubagentManager
|
||||
|
||||
|
||||
class WaitForSubagentsTool(Tool):
|
||||
"""
|
||||
Tool to wait for child subagents to complete and collect their results.
|
||||
|
||||
Use this after spawning multiple subagents to wait for all of them
|
||||
and get their results for synthesis.
|
||||
"""
|
||||
|
||||
def __init__(self, manager: "SubagentManager"):
|
||||
self._manager = manager
|
||||
|
||||
@property
|
||||
def name(self) -> str:
|
||||
return "wait_for_subagents"
|
||||
|
||||
@property
|
||||
def description(self) -> str:
|
||||
return (
|
||||
"Wait for one or more child subagents to complete and return their results. "
|
||||
"Use this after spawning subagents to collect all results before synthesizing. "
|
||||
"Blocks until all specified subagents finish."
|
||||
)
|
||||
|
||||
@property
|
||||
def parameters(self) -> dict[str, Any]:
|
||||
return {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"task_ids": {
|
||||
"type": "array",
|
||||
"items": {"type": "string"},
|
||||
"description": "List of task IDs to wait for (from spawn tool responses)",
|
||||
},
|
||||
},
|
||||
"required": ["task_ids"],
|
||||
}
|
||||
|
||||
async def execute(self, task_ids: list[str], **kwargs: Any) -> str:
|
||||
"""Wait for the specified subagents and return their results."""
|
||||
return await self._manager.wait_for(task_ids)
|
||||
@@ -0,0 +1,82 @@
|
||||
# nanobot/agent/visibility.py
|
||||
"""Cryptographic signing for visibility markers to prevent model forgery."""
|
||||
|
||||
import hmac
|
||||
import hashlib
|
||||
import re
|
||||
|
||||
SECRET_KEY = "nanobot_visibility_secret_key_v1"
|
||||
|
||||
|
||||
def compute_signature(content: str) -> str:
|
||||
"""Compute HMAC signature for content (hex string, no prefix)."""
|
||||
return hmac.new(
|
||||
SECRET_KEY.encode(),
|
||||
content.encode(),
|
||||
hashlib.sha256
|
||||
).hexdigest()[:8]
|
||||
|
||||
|
||||
def sign_content(content: str) -> str:
|
||||
"""
|
||||
Sign content with HMAC and prepend marker.
|
||||
|
||||
Args:
|
||||
content: The message content to sign
|
||||
|
||||
Returns:
|
||||
Content with signed visibility marker: "[HIDDEN:{sig}] {content}"
|
||||
"""
|
||||
sig = compute_signature(content)
|
||||
return f"[HIDDEN:{sig}] {content}"
|
||||
|
||||
|
||||
def verify_signature(marked_content: str) -> tuple[bool, str]:
|
||||
"""
|
||||
Verify HMAC signature and extract clean content.
|
||||
|
||||
Args:
|
||||
marked_content: Content potentially with [HIDDEN:{sig}] marker
|
||||
|
||||
Returns:
|
||||
Tuple of (is_valid, clean_content)
|
||||
- is_valid: True if signature is valid, False otherwise
|
||||
- clean_content: Content without marker
|
||||
"""
|
||||
match = re.match(r'\[HIDDEN:([a-f0-9]{8})\] (.*)', marked_content, re.DOTALL)
|
||||
if not match:
|
||||
return False, marked_content
|
||||
|
||||
claimed_sig, content = match.groups()
|
||||
expected_sig = compute_signature(content)
|
||||
is_valid = hmac.compare_digest(claimed_sig, expected_sig)
|
||||
return is_valid, content
|
||||
|
||||
|
||||
def has_forged_marker(content: str) -> bool:
|
||||
"""
|
||||
Check if content has an invalid [HIDDEN:*] marker at the start.
|
||||
|
||||
Args:
|
||||
content: Content to check
|
||||
|
||||
Returns:
|
||||
True if content starts with forged marker, False otherwise
|
||||
"""
|
||||
if not content.startswith("[HIDDEN:"):
|
||||
return False
|
||||
is_valid, _ = verify_signature(content)
|
||||
return not is_valid
|
||||
|
||||
|
||||
def strip_all_hidden_markers(content: str) -> str:
|
||||
"""
|
||||
Remove all [HIDDEN:*] patterns from content (valid or invalid).
|
||||
|
||||
Args:
|
||||
content: Content potentially with markers
|
||||
|
||||
Returns:
|
||||
Content with all markers stripped
|
||||
"""
|
||||
return re.sub(r'\[HIDDEN:[a-f0-9]{8}\]\s*', '', content)
|
||||
@@ -1,6 +1,7 @@
|
||||
"""Async message queue for decoupled channel-agent communication."""
|
||||
|
||||
import asyncio
|
||||
from typing import Awaitable, Callable
|
||||
|
||||
from nanobot.bus.events import InboundMessage, OutboundMessage
|
||||
|
||||
@@ -16,6 +17,9 @@ class MessageBus:
|
||||
def __init__(self):
|
||||
self.inbound: asyncio.Queue[InboundMessage] = asyncio.Queue()
|
||||
self.outbound: asyncio.Queue[OutboundMessage] = asyncio.Queue()
|
||||
self._outbound_subscribers: dict[str, list[Callable[[OutboundMessage], Awaitable[None]]]] = {}
|
||||
self._correlation_store: dict[str, asyncio.Future] = {}
|
||||
self._running = False
|
||||
|
||||
async def publish_inbound(self, msg: InboundMessage) -> None:
|
||||
"""Publish a message from a channel to the agent."""
|
||||
@@ -33,6 +37,59 @@ class MessageBus:
|
||||
"""Consume the next outbound message (blocks until available)."""
|
||||
return await self.outbound.get()
|
||||
|
||||
def register_correlation(self, correlation_id: str) -> asyncio.Future:
|
||||
"""Register a Future to be resolved when a matching outbound message appears."""
|
||||
loop = asyncio.get_running_loop()
|
||||
future = loop.create_future()
|
||||
self._correlation_store[correlation_id] = future
|
||||
return future
|
||||
|
||||
def resolve_correlation(self, msg: OutboundMessage) -> None:
|
||||
"""Check if an outbound message has a correlation_id and resolve the matching Future."""
|
||||
cid = msg.metadata.get("correlation_id") if msg.metadata else None
|
||||
if cid and cid in self._correlation_store:
|
||||
future = self._correlation_store.pop(cid)
|
||||
if not future.done():
|
||||
future.set_result(msg.content)
|
||||
|
||||
def cancel_correlation(self, correlation_id: str) -> None:
|
||||
"""Cancel and remove a pending correlation."""
|
||||
future = self._correlation_store.pop(correlation_id, None)
|
||||
if future and not future.done():
|
||||
future.cancel()
|
||||
|
||||
def subscribe_outbound(
|
||||
self,
|
||||
channel: str,
|
||||
callback: Callable[[OutboundMessage], Awaitable[None]]
|
||||
) -> None:
|
||||
"""Subscribe to outbound messages for a specific channel."""
|
||||
if channel not in self._outbound_subscribers:
|
||||
self._outbound_subscribers[channel] = []
|
||||
self._outbound_subscribers[channel].append(callback)
|
||||
|
||||
async def dispatch_outbound(self) -> None:
|
||||
"""
|
||||
Dispatch outbound messages to subscribed channels.
|
||||
Run this as a background task.
|
||||
"""
|
||||
self._running = True
|
||||
while self._running:
|
||||
try:
|
||||
msg = await asyncio.wait_for(self.outbound.get(), timeout=1.0)
|
||||
subscribers = self._outbound_subscribers.get(msg.channel, [])
|
||||
for callback in subscribers:
|
||||
try:
|
||||
await callback(msg)
|
||||
except Exception as e:
|
||||
logger.error(f"Error dispatching to {msg.channel}: {e}")
|
||||
except asyncio.TimeoutError:
|
||||
continue
|
||||
|
||||
def stop(self) -> None:
|
||||
"""Stop the dispatcher loop."""
|
||||
self._running = False
|
||||
|
||||
@property
|
||||
def inbound_size(self) -> int:
|
||||
"""Number of pending inbound messages."""
|
||||
|
||||
@@ -69,11 +69,15 @@ class BaseChannel(ABC):
|
||||
True if allowed, False otherwise.
|
||||
"""
|
||||
allow_list = getattr(self.config, "allow_from", [])
|
||||
|
||||
|
||||
# If no allow list, allow everyone
|
||||
if not allow_list:
|
||||
return True
|
||||
|
||||
|
||||
# Wildcard allows everyone
|
||||
if "*" in allow_list:
|
||||
return True
|
||||
|
||||
sender_str = str(sender_id)
|
||||
if sender_str in allow_list:
|
||||
return True
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
"""Hook channel — receives outbound messages from hook-initiated conversations."""
|
||||
|
||||
from loguru import logger
|
||||
|
||||
from nanobot.bus.events import OutboundMessage
|
||||
from nanobot.bus.queue import MessageBus
|
||||
|
||||
|
||||
class HookChannel:
|
||||
"""
|
||||
Minimal channel for hook-initiated conversations.
|
||||
|
||||
The hook HTTP server publishes InboundMessages to the bus.
|
||||
Responses come back as OutboundMessages routed here.
|
||||
send() is a no-op because the HTTP caller gets the response
|
||||
via bus correlation, not channel delivery.
|
||||
"""
|
||||
|
||||
name = "hook"
|
||||
|
||||
def __init__(self, bus: MessageBus):
|
||||
self.bus = bus
|
||||
self._running = False
|
||||
|
||||
async def start(self) -> None:
|
||||
self._running = True
|
||||
logger.info("Hook channel started")
|
||||
|
||||
async def stop(self) -> None:
|
||||
self._running = False
|
||||
|
||||
async def send(self, msg: OutboundMessage) -> None:
|
||||
"""No-op — response is returned via bus correlation to the HTTP caller."""
|
||||
logger.debug(f"Hook channel received outbound for {msg.chat_id} (no-op)")
|
||||
|
||||
@property
|
||||
def is_running(self) -> bool:
|
||||
return self._running
|
||||
@@ -149,6 +149,11 @@ class ChannelManager:
|
||||
except ImportError as e:
|
||||
logger.warning("Matrix channel not available: {}", e)
|
||||
|
||||
def register_channel(self, name: str, channel: BaseChannel) -> None:
|
||||
"""Register an external channel."""
|
||||
self.channels[name] = channel
|
||||
logger.info(f"{name} channel registered")
|
||||
|
||||
async def _start_channel(self, name: str, channel: BaseChannel) -> None:
|
||||
"""Start a channel and log any exceptions."""
|
||||
try:
|
||||
@@ -204,13 +209,16 @@ class ChannelManager:
|
||||
self.bus.consume_outbound(),
|
||||
timeout=1.0
|
||||
)
|
||||
|
||||
|
||||
# Resolve any pending correlation (hook request-response)
|
||||
self.bus.resolve_correlation(msg)
|
||||
|
||||
if msg.metadata.get("_progress"):
|
||||
if msg.metadata.get("_tool_hint") and not self.config.channels.send_tool_hints:
|
||||
continue
|
||||
if not msg.metadata.get("_tool_hint") and not self.config.channels.send_progress:
|
||||
continue
|
||||
|
||||
|
||||
channel = self.channels.get(msg.channel)
|
||||
if channel:
|
||||
try:
|
||||
|
||||
+244
-160
@@ -4,8 +4,10 @@ from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
from loguru import logger
|
||||
from telegram import BotCommand, Update, ReplyParameters
|
||||
from telegram import BotCommand, Update
|
||||
from telegram.ext import Application, CommandHandler, MessageHandler, filters, ContextTypes
|
||||
from telegram.request import HTTPXRequest
|
||||
|
||||
@@ -78,26 +80,6 @@ def _markdown_to_telegram_html(text: str) -> str:
|
||||
return text
|
||||
|
||||
|
||||
def _split_message(content: str, max_len: int = 4000) -> list[str]:
|
||||
"""Split content into chunks within max_len, preferring line breaks."""
|
||||
if len(content) <= max_len:
|
||||
return [content]
|
||||
chunks: list[str] = []
|
||||
while content:
|
||||
if len(content) <= max_len:
|
||||
chunks.append(content)
|
||||
break
|
||||
cut = content[:max_len]
|
||||
pos = cut.rfind('\n')
|
||||
if pos == -1:
|
||||
pos = cut.rfind(' ')
|
||||
if pos == -1:
|
||||
pos = max_len
|
||||
chunks.append(content[:pos])
|
||||
content = content[pos:].lstrip()
|
||||
return chunks
|
||||
|
||||
|
||||
class TelegramChannel(BaseChannel):
|
||||
"""
|
||||
Telegram channel using long polling.
|
||||
@@ -111,8 +93,8 @@ class TelegramChannel(BaseChannel):
|
||||
BOT_COMMANDS = [
|
||||
BotCommand("start", "Start the bot"),
|
||||
BotCommand("new", "Start a new conversation"),
|
||||
BotCommand("stop", "Stop the current task"),
|
||||
BotCommand("help", "Show available commands"),
|
||||
BotCommand("quota", "Show current quota status"),
|
||||
]
|
||||
|
||||
def __init__(
|
||||
@@ -127,8 +109,6 @@ class TelegramChannel(BaseChannel):
|
||||
self._app: Application | None = None
|
||||
self._chat_ids: dict[str, int] = {} # Map sender_id to chat_id for replies
|
||||
self._typing_tasks: dict[str, asyncio.Task] = {} # chat_id -> typing loop task
|
||||
self._media_group_buffers: dict[str, dict] = {}
|
||||
self._media_group_tasks: dict[str, asyncio.Task] = {}
|
||||
|
||||
async def start(self) -> None:
|
||||
"""Start the Telegram bot with long polling."""
|
||||
@@ -149,7 +129,8 @@ class TelegramChannel(BaseChannel):
|
||||
# Add command handlers
|
||||
self._app.add_handler(CommandHandler("start", self._on_start))
|
||||
self._app.add_handler(CommandHandler("new", self._forward_command))
|
||||
self._app.add_handler(CommandHandler("help", self._on_help))
|
||||
self._app.add_handler(CommandHandler("help", self._forward_command))
|
||||
self._app.add_handler(CommandHandler("quota", self._forward_command))
|
||||
|
||||
# Add message handler for text, photos, voice, documents
|
||||
self._app.add_handler(
|
||||
@@ -168,13 +149,13 @@ class TelegramChannel(BaseChannel):
|
||||
|
||||
# Get bot info and register command menu
|
||||
bot_info = await self._app.bot.get_me()
|
||||
logger.info("Telegram bot @{} connected", bot_info.username)
|
||||
logger.info(f"Telegram bot @{bot_info.username} connected")
|
||||
|
||||
try:
|
||||
await self._app.bot.set_my_commands(self.BOT_COMMANDS)
|
||||
logger.debug("Telegram bot commands registered")
|
||||
except Exception as e:
|
||||
logger.warning("Failed to register bot commands: {}", e)
|
||||
logger.warning(f"Failed to register bot commands: {e}")
|
||||
|
||||
# Start polling (this runs until stopped)
|
||||
await self._app.updater.start_polling(
|
||||
@@ -193,11 +174,6 @@ class TelegramChannel(BaseChannel):
|
||||
# Cancel all typing indicators
|
||||
for chat_id in list(self._typing_tasks):
|
||||
self._stop_typing(chat_id)
|
||||
|
||||
for task in self._media_group_tasks.values():
|
||||
task.cancel()
|
||||
self._media_group_tasks.clear()
|
||||
self._media_group_buffers.clear()
|
||||
|
||||
if self._app:
|
||||
logger.info("Stopping Telegram bot...")
|
||||
@@ -206,123 +182,264 @@ class TelegramChannel(BaseChannel):
|
||||
await self._app.shutdown()
|
||||
self._app = None
|
||||
|
||||
@staticmethod
|
||||
def _get_media_type(path: str) -> str:
|
||||
"""Guess media type from file extension."""
|
||||
ext = path.rsplit(".", 1)[-1].lower() if "." in path else ""
|
||||
if ext in ("jpg", "jpeg", "png", "gif", "webp"):
|
||||
return "photo"
|
||||
if ext == "ogg":
|
||||
return "voice"
|
||||
if ext in ("mp3", "m4a", "wav", "aac"):
|
||||
return "audio"
|
||||
return "document"
|
||||
|
||||
async def send(self, msg: OutboundMessage) -> None:
|
||||
"""Send a message through Telegram."""
|
||||
if not self._app:
|
||||
logger.warning("Telegram bot not running")
|
||||
return
|
||||
|
||||
# Stop typing indicator for this chat
|
||||
self._stop_typing(msg.chat_id)
|
||||
|
||||
# Check for suppression
|
||||
if msg.metadata.get("suppressed", False):
|
||||
logger.debug(f"Suppressed output (not sent to Telegram): {msg.content[:100]}...")
|
||||
return # Don't send to Telegram API
|
||||
|
||||
try:
|
||||
# chat_id should be the Telegram chat ID (integer)
|
||||
chat_id = int(msg.chat_id)
|
||||
# Convert markdown to Telegram HTML
|
||||
html_content = _markdown_to_telegram_html(msg.content)
|
||||
|
||||
# Check if message has media attachments
|
||||
if msg.media:
|
||||
await self._send_with_media(chat_id, html_content, msg.media)
|
||||
else:
|
||||
# Text-only message - split if too long
|
||||
await self._send_text_chunks(chat_id, html_content, parse_mode="HTML")
|
||||
except ValueError:
|
||||
logger.error("Invalid chat_id: {}", msg.chat_id)
|
||||
logger.error(f"Invalid chat_id: {msg.chat_id}")
|
||||
except Exception as e:
|
||||
# Fallback to plain text if HTML parsing fails
|
||||
logger.warning(f"HTML parse failed, falling back to plain text: {e}")
|
||||
try:
|
||||
await self._send_text_chunks(int(msg.chat_id), msg.content, parse_mode=None)
|
||||
except Exception as e2:
|
||||
logger.error(f"Error sending Telegram message: {e2}")
|
||||
|
||||
@staticmethod
|
||||
def _split_message(content: str, max_len: int = 4000) -> list[str]:
|
||||
"""Split content into chunks within max_len, preferring line breaks.
|
||||
|
||||
From upstream HKUDS/nanobot - battle-tested implementation.
|
||||
Uses 4000 char limit (safer than 4096) with split priority: \n → space → hard cut.
|
||||
"""
|
||||
if len(content) <= max_len:
|
||||
return [content]
|
||||
chunks: list[str] = []
|
||||
while content:
|
||||
if len(content) <= max_len:
|
||||
chunks.append(content)
|
||||
break
|
||||
cut = content[:max_len]
|
||||
pos = cut.rfind('\n')
|
||||
if pos == -1:
|
||||
pos = cut.rfind(' ')
|
||||
if pos == -1:
|
||||
pos = max_len
|
||||
chunks.append(content[:pos])
|
||||
content = content[pos:].lstrip()
|
||||
return chunks
|
||||
|
||||
async def _send_text_chunks(
|
||||
self,
|
||||
chat_id: int,
|
||||
text: str,
|
||||
parse_mode: str | None = "HTML"
|
||||
) -> None:
|
||||
"""Split and send long messages.
|
||||
|
||||
Telegram has a 4096 character limit per message.
|
||||
Uses upstream's proven implementation - splits at line breaks, then spaces.
|
||||
"""
|
||||
chunks = self._split_message(text)
|
||||
|
||||
for chunk in chunks:
|
||||
await self._app.bot.send_message(
|
||||
chat_id=chat_id,
|
||||
text=chunk.strip(),
|
||||
parse_mode=parse_mode
|
||||
)
|
||||
|
||||
async def _send_with_media(self, chat_id: int, caption: str, media_paths: list[str]) -> None:
|
||||
"""
|
||||
Send message with media attachments.
|
||||
|
||||
Args:
|
||||
chat_id: Telegram chat ID
|
||||
caption: Message caption
|
||||
media_paths: List of file paths or URLs
|
||||
"""
|
||||
from telegram import InputMediaPhoto, InputMediaVideo
|
||||
|
||||
from nanobot.channels.telegram_media import (
|
||||
MediaKind,
|
||||
classify_media,
|
||||
detect_mime,
|
||||
fetch_media,
|
||||
group_media_for_album,
|
||||
optimize_image,
|
||||
)
|
||||
|
||||
# Process each media item
|
||||
processed_media: list[tuple[str, MediaKind, bytes, str]] = []
|
||||
|
||||
for path in media_paths:
|
||||
try:
|
||||
# Fetch remote URLs
|
||||
if path.startswith(("http://", "https://")):
|
||||
content, mime = await fetch_media(path, max_bytes=100_000_000)
|
||||
kind = classify_media(mime)
|
||||
# Extract filename from URL
|
||||
filename = Path(path).name
|
||||
else:
|
||||
# Local file
|
||||
file_path = Path(path)
|
||||
if not file_path.exists():
|
||||
logger.warning(f"Media file not found: {path}")
|
||||
continue
|
||||
|
||||
with open(file_path, "rb") as f:
|
||||
content = f.read()
|
||||
|
||||
mime = detect_mime(path, content)
|
||||
kind = classify_media(mime)
|
||||
# Extract filename from local path
|
||||
filename = file_path.name
|
||||
|
||||
# Optimize images
|
||||
if kind == MediaKind.IMAGE:
|
||||
try:
|
||||
content = optimize_image(path, max_bytes=6_000_000)
|
||||
except Exception as e:
|
||||
logger.warning(f"Image optimization failed: {e}, sending original")
|
||||
|
||||
processed_media.append((path, kind, content, filename))
|
||||
|
||||
except Exception as e:
|
||||
logger.error(f"Failed to process media {path}: {e}")
|
||||
continue
|
||||
|
||||
if not processed_media:
|
||||
# No media could be processed, send text only
|
||||
await self._app.bot.send_message(
|
||||
chat_id=chat_id,
|
||||
text=caption,
|
||||
parse_mode="HTML"
|
||||
)
|
||||
return
|
||||
|
||||
reply_params = None
|
||||
if self.config.reply_to_message:
|
||||
reply_to_message_id = msg.metadata.get("message_id")
|
||||
if reply_to_message_id:
|
||||
reply_params = ReplyParameters(
|
||||
message_id=reply_to_message_id,
|
||||
allow_sending_without_reply=True
|
||||
)
|
||||
# Group media for album sending
|
||||
media_items = [(path, kind) for path, kind, _, _ in processed_media]
|
||||
grouping = group_media_for_album(media_items)
|
||||
|
||||
# Send media files
|
||||
for media_path in (msg.media or []):
|
||||
try:
|
||||
media_type = self._get_media_type(media_path)
|
||||
sender = {
|
||||
"photo": self._app.bot.send_photo,
|
||||
"voice": self._app.bot.send_voice,
|
||||
"audio": self._app.bot.send_audio,
|
||||
}.get(media_type, self._app.bot.send_document)
|
||||
param = "photo" if media_type == "photo" else media_type if media_type in ("voice", "audio") else "document"
|
||||
with open(media_path, 'rb') as f:
|
||||
await sender(
|
||||
chat_id=chat_id,
|
||||
**{param: f},
|
||||
reply_parameters=reply_params
|
||||
# Handle caption length (Telegram limit: 1024 chars)
|
||||
if len(caption) > 1024:
|
||||
# Send media without caption, then follow-up text
|
||||
media_caption = None
|
||||
followup_text = caption
|
||||
else:
|
||||
media_caption = caption
|
||||
followup_text = None
|
||||
|
||||
# Send album if grouped
|
||||
if grouping["album"]:
|
||||
album_paths = grouping["album"]
|
||||
album_media = []
|
||||
|
||||
for path, kind, content, filename in processed_media:
|
||||
if path not in album_paths:
|
||||
continue
|
||||
|
||||
if kind == MediaKind.IMAGE:
|
||||
media_obj = InputMediaPhoto(
|
||||
media=content,
|
||||
caption=media_caption if len(album_media) == 0 else None,
|
||||
parse_mode="HTML" if media_caption else None
|
||||
)
|
||||
except Exception as e:
|
||||
filename = media_path.rsplit("/", 1)[-1]
|
||||
logger.error("Failed to send media {}: {}", media_path, e)
|
||||
await self._app.bot.send_message(
|
||||
elif kind == MediaKind.VIDEO:
|
||||
media_obj = InputMediaVideo(
|
||||
media=content,
|
||||
caption=media_caption if len(album_media) == 0 else None,
|
||||
parse_mode="HTML" if media_caption else None
|
||||
)
|
||||
else:
|
||||
continue # Skip non-album types
|
||||
|
||||
album_media.append(media_obj)
|
||||
|
||||
if album_media:
|
||||
await self._app.bot.send_media_group(
|
||||
chat_id=chat_id,
|
||||
text=f"[Failed to send: {filename}]",
|
||||
reply_parameters=reply_params
|
||||
media=album_media
|
||||
)
|
||||
|
||||
# Send text content
|
||||
if msg.content and msg.content != "[empty message]":
|
||||
for chunk in _split_message(msg.content):
|
||||
try:
|
||||
html = _markdown_to_telegram_html(chunk)
|
||||
await self._app.bot.send_message(
|
||||
chat_id=chat_id,
|
||||
text=html,
|
||||
parse_mode="HTML",
|
||||
reply_parameters=reply_params
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning("HTML parse failed, falling back to plain text: {}", e)
|
||||
try:
|
||||
await self._app.bot.send_message(
|
||||
chat_id=chat_id,
|
||||
text=chunk,
|
||||
reply_parameters=reply_params
|
||||
)
|
||||
except Exception as e2:
|
||||
logger.error("Error sending Telegram message: {}", e2)
|
||||
|
||||
# Send separate media
|
||||
for i, (path, kind, content, filename) in enumerate(processed_media):
|
||||
if path in grouping["album"]:
|
||||
continue # Already sent in album
|
||||
|
||||
# Only first separate item gets caption
|
||||
item_caption = media_caption if i == 0 else None
|
||||
|
||||
if kind == MediaKind.IMAGE:
|
||||
await self._app.bot.send_photo(
|
||||
chat_id=chat_id,
|
||||
photo=content,
|
||||
caption=item_caption,
|
||||
parse_mode="HTML" if item_caption else None
|
||||
)
|
||||
elif kind == MediaKind.VIDEO:
|
||||
await self._app.bot.send_video(
|
||||
chat_id=chat_id,
|
||||
video=content,
|
||||
caption=item_caption,
|
||||
parse_mode="HTML" if item_caption else None
|
||||
)
|
||||
elif kind == MediaKind.AUDIO:
|
||||
await self._app.bot.send_audio(
|
||||
chat_id=chat_id,
|
||||
audio=content,
|
||||
caption=item_caption,
|
||||
parse_mode="HTML" if item_caption else None,
|
||||
filename=filename
|
||||
)
|
||||
elif kind == MediaKind.DOCUMENT:
|
||||
await self._app.bot.send_document(
|
||||
chat_id=chat_id,
|
||||
document=content,
|
||||
caption=item_caption,
|
||||
parse_mode="HTML" if item_caption else None,
|
||||
filename=filename
|
||||
)
|
||||
|
||||
# Send follow-up text if caption was too long
|
||||
if followup_text:
|
||||
await self._app.bot.send_message(
|
||||
chat_id=chat_id,
|
||||
text=followup_text,
|
||||
parse_mode="HTML"
|
||||
)
|
||||
|
||||
async def _on_start(self, update: Update, context: ContextTypes.DEFAULT_TYPE) -> None:
|
||||
"""Handle /start command."""
|
||||
if not update.message or not update.effective_user:
|
||||
return
|
||||
|
||||
|
||||
user = update.effective_user
|
||||
await update.message.reply_text(
|
||||
f"👋 Hi {user.first_name}! I'm nanobot.\n\n"
|
||||
"Send me a message and I'll respond!\n"
|
||||
"Type /help to see available commands."
|
||||
)
|
||||
|
||||
async def _on_help(self, update: Update, context: ContextTypes.DEFAULT_TYPE) -> None:
|
||||
"""Handle /help command, bypassing ACL so all users can access it."""
|
||||
if not update.message:
|
||||
return
|
||||
await update.message.reply_text(
|
||||
"🐈 nanobot commands:\n"
|
||||
"/new — Start a new conversation\n"
|
||||
"/stop — Stop the current task\n"
|
||||
"/help — Show available commands"
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _sender_id(user) -> str:
|
||||
"""Build sender_id with username for allowlist matching."""
|
||||
sid = str(user.id)
|
||||
return f"{sid}|{user.username}" if user.username else sid
|
||||
|
||||
|
||||
async def _forward_command(self, update: Update, context: ContextTypes.DEFAULT_TYPE) -> None:
|
||||
"""Forward slash commands to the bus for unified handling in AgentLoop."""
|
||||
if not update.message or not update.effective_user:
|
||||
return
|
||||
await self._handle_message(
|
||||
sender_id=self._sender_id(update.effective_user),
|
||||
sender_id=str(update.effective_user.id),
|
||||
chat_id=str(update.message.chat_id),
|
||||
content=update.message.text,
|
||||
)
|
||||
@@ -335,7 +452,11 @@ class TelegramChannel(BaseChannel):
|
||||
message = update.message
|
||||
user = update.effective_user
|
||||
chat_id = message.chat_id
|
||||
sender_id = self._sender_id(user)
|
||||
|
||||
# Use stable numeric ID, but keep username for allowlist compatibility
|
||||
sender_id = str(user.id)
|
||||
if user.username:
|
||||
sender_id = f"{sender_id}|{user.username}"
|
||||
|
||||
# Store chat_id for replies
|
||||
self._chat_ids[sender_id] = chat_id
|
||||
@@ -389,45 +510,23 @@ class TelegramChannel(BaseChannel):
|
||||
transcriber = GroqTranscriptionProvider(api_key=self.groq_api_key)
|
||||
transcription = await transcriber.transcribe(file_path)
|
||||
if transcription:
|
||||
logger.info("Transcribed {}: {}...", media_type, transcription[:50])
|
||||
logger.info(f"Transcribed {media_type}: {transcription[:50]}...")
|
||||
content_parts.append(f"[transcription: {transcription}]")
|
||||
else:
|
||||
content_parts.append(f"[{media_type}: {file_path}]")
|
||||
else:
|
||||
content_parts.append(f"[{media_type}: {file_path}]")
|
||||
|
||||
logger.debug("Downloaded {} to {}", media_type, file_path)
|
||||
logger.debug(f"Downloaded {media_type} to {file_path}")
|
||||
except Exception as e:
|
||||
logger.error("Failed to download media: {}", e)
|
||||
logger.error(f"Failed to download media: {e}")
|
||||
content_parts.append(f"[{media_type}: download failed]")
|
||||
|
||||
content = "\n".join(content_parts) if content_parts else "[empty message]"
|
||||
|
||||
logger.debug("Telegram message from {}: {}...", sender_id, content[:50])
|
||||
logger.debug(f"Telegram message from {sender_id}: {content[:50]}...")
|
||||
|
||||
str_chat_id = str(chat_id)
|
||||
|
||||
# Telegram media groups: buffer briefly, forward as one aggregated turn.
|
||||
if media_group_id := getattr(message, "media_group_id", None):
|
||||
key = f"{str_chat_id}:{media_group_id}"
|
||||
if key not in self._media_group_buffers:
|
||||
self._media_group_buffers[key] = {
|
||||
"sender_id": sender_id, "chat_id": str_chat_id,
|
||||
"contents": [], "media": [],
|
||||
"metadata": {
|
||||
"message_id": message.message_id, "user_id": user.id,
|
||||
"username": user.username, "first_name": user.first_name,
|
||||
"is_group": message.chat.type != "private",
|
||||
},
|
||||
}
|
||||
self._start_typing(str_chat_id)
|
||||
buf = self._media_group_buffers[key]
|
||||
if content and content != "[empty message]":
|
||||
buf["contents"].append(content)
|
||||
buf["media"].extend(media_paths)
|
||||
if key not in self._media_group_tasks:
|
||||
self._media_group_tasks[key] = asyncio.create_task(self._flush_media_group(key))
|
||||
return
|
||||
|
||||
# Start typing indicator before processing
|
||||
self._start_typing(str_chat_id)
|
||||
@@ -447,21 +546,6 @@ class TelegramChannel(BaseChannel):
|
||||
}
|
||||
)
|
||||
|
||||
async def _flush_media_group(self, key: str) -> None:
|
||||
"""Wait briefly, then forward buffered media-group as one turn."""
|
||||
try:
|
||||
await asyncio.sleep(0.6)
|
||||
if not (buf := self._media_group_buffers.pop(key, None)):
|
||||
return
|
||||
content = "\n".join(buf["contents"]) or "[empty message]"
|
||||
await self._handle_message(
|
||||
sender_id=buf["sender_id"], chat_id=buf["chat_id"],
|
||||
content=content, media=list(dict.fromkeys(buf["media"])),
|
||||
metadata=buf["metadata"],
|
||||
)
|
||||
finally:
|
||||
self._media_group_tasks.pop(key, None)
|
||||
|
||||
def _start_typing(self, chat_id: str) -> None:
|
||||
"""Start sending 'typing...' indicator for a chat."""
|
||||
# Cancel any existing typing task for this chat
|
||||
@@ -483,11 +567,11 @@ class TelegramChannel(BaseChannel):
|
||||
except asyncio.CancelledError:
|
||||
pass
|
||||
except Exception as e:
|
||||
logger.debug("Typing indicator stopped for {}: {}", chat_id, e)
|
||||
logger.debug(f"Typing indicator stopped for {chat_id}: {e}")
|
||||
|
||||
async def _on_error(self, update: object, context: ContextTypes.DEFAULT_TYPE) -> None:
|
||||
"""Log polling / handler errors instead of silently swallowing them."""
|
||||
logger.error("Telegram error: {}", context.error)
|
||||
logger.error(f"Telegram error: {context.error}")
|
||||
|
||||
def _get_extension(self, media_type: str, mime_type: str | None) -> str:
|
||||
"""Get file extension based on media type."""
|
||||
|
||||
@@ -0,0 +1,286 @@
|
||||
"""Media handling utilities for Telegram channel."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import mimetypes
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
|
||||
import httpx
|
||||
from loguru import logger
|
||||
from PIL import Image
|
||||
|
||||
# Telegram API photo size limit (6MB)
|
||||
TELEGRAM_PHOTO_SIZE_LIMIT = 6_000_000
|
||||
|
||||
try:
|
||||
import magic
|
||||
HAS_MAGIC = True
|
||||
except ImportError:
|
||||
HAS_MAGIC = False
|
||||
|
||||
try:
|
||||
from pillow_heif import register_heif_opener
|
||||
register_heif_opener()
|
||||
HAS_HEIF = True
|
||||
except ImportError:
|
||||
HAS_HEIF = False
|
||||
|
||||
|
||||
class MediaKind(Enum):
|
||||
"""Media type classification."""
|
||||
IMAGE = "image"
|
||||
VIDEO = "video"
|
||||
AUDIO = "audio"
|
||||
DOCUMENT = "document"
|
||||
|
||||
|
||||
def detect_mime(path: str, content: bytes | None = None) -> str:
|
||||
"""
|
||||
Detect MIME type of media file.
|
||||
|
||||
Priority:
|
||||
1. python-magic sniff (if available and content provided)
|
||||
2. Extension-based lookup
|
||||
3. Fallback to application/octet-stream
|
||||
|
||||
Args:
|
||||
path: File path (used for extension detection)
|
||||
content: Optional file content bytes for magic sniffing
|
||||
|
||||
Returns:
|
||||
MIME type string (e.g., "image/jpeg")
|
||||
"""
|
||||
# Try magic detection first if we have content
|
||||
if HAS_MAGIC and content:
|
||||
try:
|
||||
mime = magic.from_buffer(content, mime=True)
|
||||
# Avoid generic types if we can be more specific from extension
|
||||
if mime and mime != "application/octet-stream":
|
||||
return mime
|
||||
except Exception as e:
|
||||
logger.debug(f"Magic detection failed, falling back to extension: {e}")
|
||||
|
||||
# Extension-based detection
|
||||
mime_type, _ = mimetypes.guess_type(path)
|
||||
if mime_type:
|
||||
return mime_type
|
||||
|
||||
# Fallback
|
||||
return "application/octet-stream"
|
||||
|
||||
|
||||
def classify_media(mime: str) -> MediaKind:
|
||||
"""
|
||||
Classify MIME type into media kind.
|
||||
|
||||
Args:
|
||||
mime: MIME type string (e.g., "image/jpeg")
|
||||
|
||||
Returns:
|
||||
MediaKind enum value
|
||||
"""
|
||||
if mime.startswith("image/"):
|
||||
return MediaKind.IMAGE
|
||||
if mime.startswith("video/"):
|
||||
return MediaKind.VIDEO
|
||||
if mime.startswith("audio/"):
|
||||
return MediaKind.AUDIO
|
||||
# Everything else is a document
|
||||
return MediaKind.DOCUMENT
|
||||
|
||||
|
||||
def is_heic_format(path: str) -> bool:
|
||||
"""
|
||||
Check if file is HEIC/HEIF format.
|
||||
|
||||
Args:
|
||||
path: File path
|
||||
|
||||
Returns:
|
||||
True if file extension is .heic or .heif
|
||||
"""
|
||||
ext = Path(path).suffix.lower()
|
||||
return ext in (".heic", ".heif")
|
||||
|
||||
|
||||
async def fetch_media(url: str, max_bytes: int) -> tuple[bytes, str]:
|
||||
"""
|
||||
Download media from remote URL.
|
||||
|
||||
Args:
|
||||
url: Remote URL to fetch
|
||||
max_bytes: Maximum size to download
|
||||
|
||||
Returns:
|
||||
Tuple of (content bytes, detected MIME type)
|
||||
|
||||
Raises:
|
||||
ValueError: If download fails or exceeds size limit
|
||||
"""
|
||||
try:
|
||||
async with httpx.AsyncClient(timeout=10.0) as client:
|
||||
response = await client.get(url, follow_redirects=True)
|
||||
response.raise_for_status()
|
||||
|
||||
content = response.content
|
||||
|
||||
if len(content) > max_bytes:
|
||||
raise ValueError(f"Media exceeds size limit: {len(content)} > {max_bytes}")
|
||||
|
||||
# Get MIME type from response or detect
|
||||
mime = response.headers.get("content-type", "application/octet-stream")
|
||||
# Strip charset if present (e.g., "image/jpeg; charset=utf-8" → "image/jpeg")
|
||||
mime = mime.split(";")[0].strip()
|
||||
|
||||
# Detect from content if generic type
|
||||
if mime == "application/octet-stream":
|
||||
mime = detect_mime(url, content)
|
||||
|
||||
return content, mime
|
||||
|
||||
except httpx.TimeoutException as e:
|
||||
raise ValueError(f"Download timeout: {url}") from e
|
||||
except httpx.HTTPError as e:
|
||||
raise ValueError(f"Download failed: {url}: {e}") from e
|
||||
|
||||
|
||||
def optimize_image(path: str, max_bytes: int = TELEGRAM_PHOTO_SIZE_LIMIT) -> bytes:
|
||||
"""
|
||||
Optimize image to fit under size limit.
|
||||
|
||||
Strategy:
|
||||
1. Convert HEIC to JPEG if needed
|
||||
2. PNG with alpha → preserve with compression levels [6,7,8,9]
|
||||
3. JPEG/PNG without alpha → resize + quality grid
|
||||
|
||||
Sizes: [2048, 1536, 1280, 1024, 800] px (max dimension)
|
||||
Qualities: [80, 70, 60, 50, 40] (JPEG only)
|
||||
|
||||
Args:
|
||||
path: Path to image file
|
||||
max_bytes: Maximum size in bytes (default 6MB for Telegram)
|
||||
|
||||
Returns:
|
||||
Optimized image bytes
|
||||
|
||||
Raises:
|
||||
ValueError: If image cannot be optimized under limit
|
||||
"""
|
||||
# Load image with context manager to ensure file handle is closed
|
||||
with Image.open(path) as img:
|
||||
# Convert HEIC to JPEG
|
||||
if is_heic_format(path):
|
||||
if not HAS_HEIF:
|
||||
raise ValueError("pillow-heif not available for HEIC conversion")
|
||||
# Convert to RGB (HEIC → JPEG)
|
||||
if img.mode != "RGB":
|
||||
img = img.convert("RGB")
|
||||
return _optimize_jpeg(img, max_bytes)
|
||||
|
||||
# PNG with alpha channel - preserve it
|
||||
if img.mode == "RGBA" or img.mode == "LA":
|
||||
return _optimize_png(img, max_bytes)
|
||||
|
||||
# Everything else → convert to JPEG and optimize
|
||||
if img.mode != "RGB":
|
||||
img = img.convert("RGB")
|
||||
return _optimize_jpeg(img, max_bytes)
|
||||
|
||||
|
||||
def _optimize_jpeg(img: Image.Image, max_bytes: int) -> bytes:
|
||||
"""Optimize JPEG with size/quality grid."""
|
||||
sizes = [2048, 1536, 1280, 1024, 800]
|
||||
qualities = [80, 70, 60, 50, 40]
|
||||
|
||||
for size in sizes:
|
||||
# Always copy to avoid mutation issues
|
||||
resized = img.copy()
|
||||
if max(img.size) > size:
|
||||
resized.thumbnail((size, size), Image.Resampling.LANCZOS)
|
||||
|
||||
for quality in qualities:
|
||||
buf = io.BytesIO()
|
||||
resized.save(buf, format="JPEG", quality=quality, optimize=True)
|
||||
data = buf.getvalue()
|
||||
|
||||
if len(data) <= max_bytes:
|
||||
return data
|
||||
|
||||
# If we get here, even smallest size/quality is too large
|
||||
raise ValueError(f"Cannot optimize image under {max_bytes} bytes")
|
||||
|
||||
|
||||
def _optimize_png(img: Image.Image, max_bytes: int) -> bytes:
|
||||
"""Optimize PNG while preserving alpha channel."""
|
||||
compress_levels = [6, 7, 8, 9]
|
||||
sizes = [2048, 1536, 1280, 1024, 800]
|
||||
|
||||
for size in sizes:
|
||||
# Always copy to avoid mutation issues
|
||||
resized = img.copy()
|
||||
if max(img.size) > size:
|
||||
resized.thumbnail((size, size), Image.Resampling.LANCZOS)
|
||||
|
||||
for compress_level in compress_levels:
|
||||
buf = io.BytesIO()
|
||||
resized.save(buf, format="PNG", compress_level=compress_level, optimize=True)
|
||||
data = buf.getvalue()
|
||||
|
||||
if len(data) <= max_bytes:
|
||||
return data
|
||||
|
||||
# Fallback: try converting to JPEG if still too large
|
||||
if img.mode in ("RGBA", "LA"):
|
||||
# Create white background
|
||||
background = Image.new("RGB", img.size, (255, 255, 255))
|
||||
if img.mode == "RGBA":
|
||||
background.paste(img, mask=img.split()[3]) # Use alpha as mask
|
||||
else: # LA (grayscale + alpha)
|
||||
background.paste(img.convert("L"), mask=img.split()[1])
|
||||
return _optimize_jpeg(background, max_bytes)
|
||||
|
||||
raise ValueError(f"Cannot optimize PNG under {max_bytes} bytes")
|
||||
|
||||
|
||||
def group_media_for_album(media_items: list[tuple[str, MediaKind]]) -> dict[str, list[str]]:
|
||||
"""
|
||||
Group media items for album sending.
|
||||
|
||||
Logic:
|
||||
- All images (2+) → album
|
||||
- All videos (2+) → album
|
||||
- Mixed types → separate
|
||||
- Single item → separate
|
||||
|
||||
Args:
|
||||
media_items: List of (path, MediaKind) tuples
|
||||
|
||||
Returns:
|
||||
Dict with 'album' and 'separate' keys containing lists of paths
|
||||
"""
|
||||
if len(media_items) <= 1:
|
||||
return {
|
||||
"album": [],
|
||||
"separate": [path for path, _ in media_items]
|
||||
}
|
||||
|
||||
# Count each kind
|
||||
kinds = [kind for _, kind in media_items]
|
||||
unique_kinds = set(kinds)
|
||||
|
||||
# All same type → album (if images or videos)
|
||||
if len(unique_kinds) == 1:
|
||||
kind = kinds[0]
|
||||
if kind in (MediaKind.IMAGE, MediaKind.VIDEO):
|
||||
return {
|
||||
"album": [path for path, _ in media_items],
|
||||
"separate": []
|
||||
}
|
||||
|
||||
# Mixed types or non-album-able types → separate
|
||||
return {
|
||||
"album": [],
|
||||
"separate": [path for path, _ in media_items]
|
||||
}
|
||||
+318
-433
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,93 @@
|
||||
"""OAuth CLI commands for subscription authentication."""
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import typer
|
||||
from rich.console import Console
|
||||
|
||||
oauth_app = typer.Typer(help="Manage OAuth authentication for subscription-based providers")
|
||||
console = Console()
|
||||
|
||||
|
||||
@oauth_app.command("login")
|
||||
def login(
|
||||
provider: str = typer.Argument("anthropic", help="Provider name"),
|
||||
token: Optional[str] = typer.Option(None, "--token", "-t", help="OAuth token (from claude setup-token)"),
|
||||
):
|
||||
"""Login to a provider using OAuth.
|
||||
|
||||
For Anthropic Claude Max/Pro, run 'claude setup-token' and paste the token here.
|
||||
|
||||
Example:
|
||||
nanobot oauth login anthropic --token sk-ant-oat01-xxx
|
||||
"""
|
||||
from nanobot.config.oauth_store import OAuthStore
|
||||
from nanobot.config.schema import OAuthCredentials
|
||||
|
||||
if provider != "anthropic":
|
||||
console.print(f"[red]OAuth login for {provider} not yet supported[/red]")
|
||||
return
|
||||
|
||||
if not token:
|
||||
console.print("Please provide your OAuth token:")
|
||||
console.print(" 1. Run: claude setup-token")
|
||||
console.print(" 2. Copy the sk-ant-oat01-... token")
|
||||
console.print(" 3. Run: nanobot oauth login anthropic --token <your-token>")
|
||||
console.print()
|
||||
token = typer.prompt("Token", hide_input=True)
|
||||
|
||||
if not token or "sk-ant-oat" not in token:
|
||||
console.print("[red]Invalid token. Must contain sk-ant-oat[/red]")
|
||||
return
|
||||
|
||||
store = OAuthStore(Path.home() / ".nanobot")
|
||||
creds = OAuthCredentials(
|
||||
access_token=token,
|
||||
token_type="token" # setup-token doesn't expire
|
||||
)
|
||||
store.save(provider, creds)
|
||||
|
||||
console.print(f"[green]Successfully saved {provider} OAuth credentials![/green]")
|
||||
|
||||
|
||||
@oauth_app.command("status")
|
||||
def status():
|
||||
"""Show OAuth credential status."""
|
||||
from nanobot.config.oauth_store import OAuthStore
|
||||
|
||||
store = OAuthStore(Path.home() / ".nanobot")
|
||||
|
||||
providers = ["anthropic"]
|
||||
found_any = False
|
||||
|
||||
for provider in providers:
|
||||
creds = store.load(provider)
|
||||
if creds:
|
||||
found_any = True
|
||||
st = "valid"
|
||||
if creds.is_expired:
|
||||
st = "EXPIRED"
|
||||
elif creds.expires_soon:
|
||||
st = "expires soon"
|
||||
|
||||
token_preview = creds.access_token[:20] + "..."
|
||||
console.print(f" {provider}: {token_preview} ({st})")
|
||||
|
||||
if not found_any:
|
||||
console.print("No OAuth credentials configured.")
|
||||
console.print("Run: nanobot oauth login anthropic --token <token>")
|
||||
|
||||
|
||||
@oauth_app.command("logout")
|
||||
def logout(
|
||||
provider: str = typer.Argument("anthropic", help="Provider name"),
|
||||
):
|
||||
"""Remove OAuth credentials for a provider."""
|
||||
from nanobot.config.oauth_store import OAuthStore
|
||||
|
||||
store = OAuthStore(Path.home() / ".nanobot")
|
||||
if store.delete(provider):
|
||||
console.print(f"[green]Removed {provider} OAuth credentials[/green]")
|
||||
else:
|
||||
console.print(f"No credentials found for {provider}")
|
||||
@@ -1,22 +1,47 @@
|
||||
"""Configuration loading utilities."""
|
||||
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
from nanobot.config.schema import Config
|
||||
|
||||
|
||||
def get_config_path() -> Path:
|
||||
"""Get the default configuration file path."""
|
||||
"""Get the configuration file path.
|
||||
|
||||
Checks NANOBOT_CONFIG environment variable first, otherwise defaults
|
||||
to ~/.nanobot/config.json
|
||||
"""
|
||||
env_path = os.getenv("NANOBOT_CONFIG")
|
||||
if env_path:
|
||||
return Path(env_path)
|
||||
return Path.home() / ".nanobot" / "config.json"
|
||||
|
||||
|
||||
def _get_oauth_store_dir() -> Path:
|
||||
"""Get the OAuth store directory."""
|
||||
return Path.home() / ".nanobot"
|
||||
|
||||
|
||||
def get_data_dir() -> Path:
|
||||
"""Get the nanobot data directory."""
|
||||
from nanobot.utils.helpers import get_data_path
|
||||
return get_data_path()
|
||||
|
||||
|
||||
def _inject_oauth_credentials(config: Config) -> Config:
|
||||
"""Inject OAuth credentials from store into config if available."""
|
||||
from nanobot.config.oauth_store import OAuthStore
|
||||
|
||||
store = OAuthStore(_get_oauth_store_dir())
|
||||
creds = store.load("anthropic")
|
||||
if creds and creds.access_token and not creds.is_expired:
|
||||
config.providers.anthropic.api_key = creds.access_token
|
||||
|
||||
return config
|
||||
|
||||
|
||||
def load_config(config_path: Path | None = None) -> Config:
|
||||
"""
|
||||
Load configuration from file or create default.
|
||||
@@ -34,12 +59,13 @@ def load_config(config_path: Path | None = None) -> Config:
|
||||
with open(path, encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
data = _migrate_config(data)
|
||||
return Config.model_validate(data)
|
||||
config = Config.model_validate(data)
|
||||
return _inject_oauth_credentials(config)
|
||||
except (json.JSONDecodeError, ValueError) as e:
|
||||
print(f"Warning: Failed to load config from {path}: {e}")
|
||||
print("Using default configuration.")
|
||||
|
||||
return Config()
|
||||
return _inject_oauth_credentials(Config())
|
||||
|
||||
|
||||
def save_config(config: Config, config_path: Path | None = None) -> None:
|
||||
@@ -66,4 +92,18 @@ def _migrate_config(data: dict) -> dict:
|
||||
exec_cfg = tools.get("exec", {})
|
||||
if "restrictToWorkspace" in exec_cfg and "restrictToWorkspace" not in tools:
|
||||
tools["restrictToWorkspace"] = exec_cfg.pop("restrictToWorkspace")
|
||||
|
||||
# Extract api_key from oauthCredentials if present
|
||||
providers = data.get("providers", {})
|
||||
for _, provider_config in providers.items():
|
||||
if isinstance(provider_config, dict):
|
||||
oauth_creds = provider_config.get("oauthCredentials")
|
||||
if oauth_creds and isinstance(oauth_creds, dict):
|
||||
access_token = oauth_creds.get("access_token", "")
|
||||
# Only set api_key if not already set and access_token exists
|
||||
if access_token and not provider_config.get("api_key"):
|
||||
provider_config["api_key"] = access_token
|
||||
# Clean up migrated data to avoid duplication
|
||||
del provider_config["oauthCredentials"]
|
||||
|
||||
return data
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
"""OAuth credential storage."""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from nanobot.config.schema import OAuthCredentials
|
||||
|
||||
|
||||
class OAuthStore:
|
||||
"""Stores OAuth credentials in a JSON file."""
|
||||
|
||||
FILENAME = "oauth-credentials.json"
|
||||
|
||||
def __init__(self, config_dir: Path):
|
||||
self.config_dir = config_dir
|
||||
self.file_path = config_dir / self.FILENAME
|
||||
|
||||
def _load_all(self) -> dict[str, Any]:
|
||||
"""Load all credentials from file."""
|
||||
if not self.file_path.exists():
|
||||
return {}
|
||||
|
||||
with open(self.file_path, "r") as f:
|
||||
return json.load(f)
|
||||
|
||||
def _save_all(self, data: dict[str, Any]) -> None:
|
||||
"""Save all credentials to file."""
|
||||
self.config_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
with open(self.file_path, "w") as f:
|
||||
json.dump(data, f, indent=2)
|
||||
|
||||
# Secure permissions
|
||||
self.file_path.chmod(0o600)
|
||||
|
||||
def save(self, provider: str, credentials: OAuthCredentials) -> None:
|
||||
"""Save credentials for a provider."""
|
||||
data = self._load_all()
|
||||
data[provider] = credentials.model_dump()
|
||||
self._save_all(data)
|
||||
|
||||
def load(self, provider: str) -> OAuthCredentials | None:
|
||||
"""Load credentials for a provider."""
|
||||
data = self._load_all()
|
||||
if provider not in data:
|
||||
return None
|
||||
|
||||
return OAuthCredentials(**data[provider])
|
||||
|
||||
def delete(self, provider: str) -> bool:
|
||||
"""Delete credentials for a provider."""
|
||||
data = self._load_all()
|
||||
if provider not in data:
|
||||
return False
|
||||
|
||||
del data[provider]
|
||||
self._save_all(data)
|
||||
return True
|
||||
@@ -1,7 +1,7 @@
|
||||
"""Configuration schema using Pydantic."""
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Literal
|
||||
from typing import Any, Literal
|
||||
|
||||
from pydantic import BaseModel, Field, ConfigDict
|
||||
from pydantic.alias_generators import to_camel
|
||||
@@ -220,12 +220,13 @@ class AgentDefaults(Base):
|
||||
"""Default agent configuration."""
|
||||
|
||||
workspace: str = "~/.nanobot/workspace"
|
||||
model: str = "anthropic/claude-opus-4-5"
|
||||
model: str = "anthropic/claude-opus-4-7"
|
||||
provider: str = "auto" # Provider name (e.g. "anthropic", "openrouter") or "auto" for auto-detection
|
||||
max_tokens: int = 8192
|
||||
temperature: float = 0.1
|
||||
max_tool_iterations: int = 40
|
||||
memory_window: int = 100
|
||||
thinking_budget: int = 0 # 0 = disabled; >0 = token budget for extended thinking
|
||||
|
||||
|
||||
class AgentsConfig(Base):
|
||||
@@ -234,12 +235,42 @@ class AgentsConfig(Base):
|
||||
defaults: AgentDefaults = Field(default_factory=AgentDefaults)
|
||||
|
||||
|
||||
class OAuthCredentials(BaseModel):
|
||||
"""OAuth token credentials for subscription-based auth."""
|
||||
access_token: str = ""
|
||||
refresh_token: str = ""
|
||||
expires_at: int = 0 # Unix timestamp
|
||||
token_type: str = "oauth" # "oauth" or "token" (setup-token)
|
||||
|
||||
@property
|
||||
def is_oauth_token(self) -> bool:
|
||||
"""Check if this is an OAuth token (vs regular API key)."""
|
||||
return "sk-ant-oat" in self.access_token
|
||||
|
||||
@property
|
||||
def is_expired(self) -> bool:
|
||||
"""Check if token has expired."""
|
||||
import time
|
||||
if self.expires_at == 0:
|
||||
return False # No expiry set (setup-token)
|
||||
return time.time() > self.expires_at
|
||||
|
||||
@property
|
||||
def expires_soon(self) -> bool:
|
||||
"""Check if token expires within 10 minutes."""
|
||||
import time
|
||||
if self.expires_at == 0:
|
||||
return False
|
||||
return time.time() > (self.expires_at - 600)
|
||||
|
||||
|
||||
class ProviderConfig(Base):
|
||||
"""LLM provider configuration."""
|
||||
|
||||
api_key: str = ""
|
||||
api_base: str | None = None
|
||||
extra_headers: dict[str, str] | None = None # Custom headers (e.g. APP-Code for AiHubMix)
|
||||
oauth_credentials: OAuthCredentials | None = None
|
||||
|
||||
|
||||
class ProvidersConfig(Base):
|
||||
@@ -279,7 +310,27 @@ class GatewayConfig(Base):
|
||||
heartbeat: HeartbeatConfig = Field(default_factory=HeartbeatConfig)
|
||||
|
||||
|
||||
class WebSearchConfig(Base):
|
||||
class HooksConfig(BaseModel):
|
||||
"""Webhook endpoint configuration."""
|
||||
enabled: bool = False
|
||||
tokens: dict[str, str] = Field(default_factory=dict) # Named tokens: {name: secret}
|
||||
path: str = "/hooks" # URL path for the endpoint
|
||||
timeout_seconds: int = 120 # Max time to wait for agent response
|
||||
|
||||
def resolve_token(self, provided: str) -> str | None:
|
||||
"""Return token name if provided secret matches, else None."""
|
||||
for name, secret in self.tokens.items():
|
||||
if secret == provided:
|
||||
return name
|
||||
return None
|
||||
|
||||
@property
|
||||
def has_tokens(self) -> bool:
|
||||
"""True if at least one token is configured."""
|
||||
return bool(self.tokens)
|
||||
|
||||
|
||||
class WebSearchConfig(BaseModel):
|
||||
"""Web search tool configuration."""
|
||||
|
||||
api_key: str = "" # Brave Search API key
|
||||
@@ -310,12 +361,25 @@ class MCPServerConfig(Base):
|
||||
tool_timeout: int = 30 # Seconds before a tool call is cancelled
|
||||
|
||||
|
||||
class Mem0Config(Base):
|
||||
"""Mem0 memory system configuration."""
|
||||
|
||||
enabled: bool = False # If true, use mem0 for semantic memory instead of simple MEMORY.md
|
||||
api_key: str = "" # Optional: mem0 cloud API key (leave empty for self-hosted)
|
||||
search_limit: int = 5 # Max memories to retrieve per query
|
||||
llm: str = "" # Optional: LLM for memory extraction (default: gpt-4.1-nano-2025-04-14)
|
||||
embedder: str = "" # Optional: Embedding model (default: mem0's default)
|
||||
vector_store: dict[str, Any] = Field(default_factory=dict) # Vector store config (e.g., {"provider": "qdrant", "config": {...}})
|
||||
|
||||
|
||||
class ToolsConfig(Base):
|
||||
"""Tools configuration."""
|
||||
|
||||
web: WebToolsConfig = Field(default_factory=WebToolsConfig)
|
||||
exec: ExecToolConfig = Field(default_factory=ExecToolConfig)
|
||||
restrict_to_workspace: bool = False # If true, restrict all tool access to workspace directory
|
||||
enable_memory_tool: bool = True # If true, enable Anthropic's native memory tool
|
||||
mem0: Mem0Config = Field(default_factory=Mem0Config) # Mem0 semantic memory configuration
|
||||
mcp_servers: dict[str, MCPServerConfig] = Field(default_factory=dict)
|
||||
|
||||
|
||||
@@ -326,6 +390,7 @@ class Config(BaseSettings):
|
||||
channels: ChannelsConfig = Field(default_factory=ChannelsConfig)
|
||||
providers: ProvidersConfig = Field(default_factory=ProvidersConfig)
|
||||
gateway: GatewayConfig = Field(default_factory=GatewayConfig)
|
||||
hooks: HooksConfig = Field(default_factory=HooksConfig)
|
||||
tools: ToolsConfig = Field(default_factory=ToolsConfig)
|
||||
|
||||
@property
|
||||
|
||||
@@ -9,64 +9,62 @@ from typing import TYPE_CHECKING, Any, Callable, Coroutine
|
||||
from loguru import logger
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from nanobot.providers.base import LLMProvider
|
||||
from nanobot.session.manager import SessionManager
|
||||
|
||||
_HEARTBEAT_TOOL = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "heartbeat",
|
||||
"description": "Report heartbeat decision after reviewing tasks.",
|
||||
"parameters": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"action": {
|
||||
"type": "string",
|
||||
"enum": ["skip", "run"],
|
||||
"description": "skip = nothing to do, run = has active tasks",
|
||||
},
|
||||
"tasks": {
|
||||
"type": "string",
|
||||
"description": "Natural-language summary of active tasks (required for run)",
|
||||
},
|
||||
},
|
||||
"required": ["action"],
|
||||
},
|
||||
},
|
||||
}
|
||||
]
|
||||
# Default interval: 30 minutes
|
||||
DEFAULT_HEARTBEAT_INTERVAL_S = 30 * 60
|
||||
|
||||
# The prompt sent to agent during heartbeat
|
||||
HEARTBEAT_PROMPT = """Read HEARTBEAT.md in your workspace (if it exists).
|
||||
Follow any instructions or tasks listed there.
|
||||
If nothing needs attention, reply with just: HEARTBEAT_OK"""
|
||||
|
||||
# Token that indicates "nothing to do"
|
||||
HEARTBEAT_OK_TOKEN = "HEARTBEAT_OK"
|
||||
|
||||
|
||||
def _is_heartbeat_empty(content: str | None) -> bool:
|
||||
"""Check if HEARTBEAT.md has no actionable content."""
|
||||
if not content:
|
||||
return True
|
||||
|
||||
# Lines to skip: empty, headers, HTML comments, empty checkboxes
|
||||
skip_patterns = {"- [ ]", "* [ ]", "- [x]", "* [x]"}
|
||||
|
||||
for line in content.split("\n"):
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#") or line.startswith("<!--") or line in skip_patterns:
|
||||
continue
|
||||
return False # Found actionable content
|
||||
|
||||
return True
|
||||
|
||||
|
||||
class HeartbeatService:
|
||||
"""
|
||||
Periodic heartbeat service that wakes the agent to check for tasks.
|
||||
|
||||
Phase 1 (decision): reads HEARTBEAT.md and asks the LLM — via a virtual
|
||||
tool call — whether there are active tasks. This avoids free-text parsing
|
||||
and the unreliable HEARTBEAT_OK token.
|
||||
|
||||
Phase 2 (execution): only triggered when Phase 1 returns ``run``. The
|
||||
``on_execute`` callback runs the task through the full agent loop and
|
||||
returns the result to deliver.
|
||||
The agent reads HEARTBEAT.md from the workspace and executes any
|
||||
tasks listed there. If nothing needs attention, it replies HEARTBEAT_OK.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
workspace: Path,
|
||||
provider: LLMProvider,
|
||||
model: str,
|
||||
on_execute: Callable[[str], Coroutine[Any, Any, str]] | None = None,
|
||||
on_notify: Callable[[str], Coroutine[Any, Any, None]] | None = None,
|
||||
interval_s: int = 30 * 60,
|
||||
on_heartbeat: Callable[[str, dict[str, Any] | None], Coroutine[Any, Any, str]] | None = None,
|
||||
interval_s: int = DEFAULT_HEARTBEAT_INTERVAL_S,
|
||||
enabled: bool = True,
|
||||
session_manager: SessionManager | None = None,
|
||||
target_session_key: str = "telegram:239824268",
|
||||
idle_threshold_s: int = 30 * 60, # 30 minutes
|
||||
):
|
||||
self.workspace = workspace
|
||||
self.provider = provider
|
||||
self.model = model
|
||||
self.on_execute = on_execute
|
||||
self.on_notify = on_notify
|
||||
self.on_heartbeat = on_heartbeat
|
||||
self.interval_s = interval_s
|
||||
self.enabled = enabled
|
||||
self.session_manager = session_manager
|
||||
self.target_session_key = target_session_key
|
||||
self.idle_threshold_s = idle_threshold_s
|
||||
self._running = False
|
||||
self._task: asyncio.Task | None = None
|
||||
|
||||
@@ -75,48 +73,27 @@ class HeartbeatService:
|
||||
return self.workspace / "HEARTBEAT.md"
|
||||
|
||||
def _read_heartbeat_file(self) -> str | None:
|
||||
"""Read HEARTBEAT.md content."""
|
||||
if self.heartbeat_file.exists():
|
||||
try:
|
||||
return self.heartbeat_file.read_text(encoding="utf-8")
|
||||
return self.heartbeat_file.read_text()
|
||||
except Exception:
|
||||
return None
|
||||
return None
|
||||
|
||||
async def _decide(self, content: str) -> tuple[str, str]:
|
||||
"""Phase 1: ask LLM to decide skip/run via virtual tool call.
|
||||
|
||||
Returns (action, tasks) where action is 'skip' or 'run'.
|
||||
"""
|
||||
response = await self.provider.chat(
|
||||
messages=[
|
||||
{"role": "system", "content": "You are a heartbeat agent. Call the heartbeat tool to report your decision."},
|
||||
{"role": "user", "content": (
|
||||
"Review the following HEARTBEAT.md and decide whether there are active tasks.\n\n"
|
||||
f"{content}"
|
||||
)},
|
||||
],
|
||||
tools=_HEARTBEAT_TOOL,
|
||||
model=self.model,
|
||||
)
|
||||
|
||||
if not response.has_tool_calls:
|
||||
return "skip", ""
|
||||
|
||||
args = response.tool_calls[0].arguments
|
||||
return args.get("action", "skip"), args.get("tasks", "")
|
||||
|
||||
async def start(self) -> None:
|
||||
"""Start the heartbeat service."""
|
||||
if not self.enabled:
|
||||
logger.info("Heartbeat disabled")
|
||||
return
|
||||
if self._running:
|
||||
logger.warning("Heartbeat already running")
|
||||
|
||||
# Idempotent: don't create a new task if already running
|
||||
if self._task is not None and not self._task.done():
|
||||
return
|
||||
|
||||
self._running = True
|
||||
self._task = asyncio.create_task(self._run_loop())
|
||||
logger.info("Heartbeat started (every {}s)", self.interval_s)
|
||||
logger.info(f"Heartbeat started (every {self.interval_s}s)")
|
||||
|
||||
def stop(self) -> None:
|
||||
"""Stop the heartbeat service."""
|
||||
@@ -135,39 +112,69 @@ class HeartbeatService:
|
||||
except asyncio.CancelledError:
|
||||
break
|
||||
except Exception as e:
|
||||
logger.error("Heartbeat error: {}", e)
|
||||
logger.error(f"Heartbeat error: {e}")
|
||||
|
||||
async def _tick(self) -> None:
|
||||
"""Execute a single heartbeat tick."""
|
||||
|
||||
# Check if user is idle (if session manager provided)
|
||||
if self.session_manager and self.target_session_key:
|
||||
try:
|
||||
session = self.session_manager.get_or_create(self.target_session_key)
|
||||
|
||||
# Find last real user message timestamp (exclude system-generated messages)
|
||||
# Real Telegram messages have sender_id like "239824268|username"
|
||||
# System messages (heartbeat, cron) created via process_direct have sender_id="user"
|
||||
# Old messages may not have sender_id field (backwards compat: treat as real user messages)
|
||||
last_user_timestamp = None
|
||||
for msg in reversed(session.messages):
|
||||
if msg.get("role") == "user":
|
||||
sender_id = msg.get("sender_id")
|
||||
# Skip if explicitly marked as system-generated
|
||||
if sender_id == "user":
|
||||
continue
|
||||
# Accept if no sender_id (old message) or if real user ID
|
||||
last_user_timestamp = msg.get("timestamp")
|
||||
break
|
||||
|
||||
if last_user_timestamp:
|
||||
from datetime import datetime
|
||||
last_dt = datetime.fromisoformat(last_user_timestamp)
|
||||
elapsed = (datetime.now() - last_dt).total_seconds()
|
||||
|
||||
if elapsed < self.idle_threshold_s:
|
||||
logger.debug(f"Heartbeat: user active {int(elapsed)}s ago, skipping")
|
||||
return # User is active, don't trigger heartbeat
|
||||
except Exception as e:
|
||||
logger.warning(f"Heartbeat: error checking idle state: {e}")
|
||||
# Continue with heartbeat on error (fail open)
|
||||
|
||||
# Original heartbeat logic
|
||||
content = self._read_heartbeat_file()
|
||||
if not content:
|
||||
logger.debug("Heartbeat: HEARTBEAT.md missing or empty")
|
||||
|
||||
# Skip if HEARTBEAT.md is empty or doesn't exist
|
||||
if _is_heartbeat_empty(content):
|
||||
logger.debug("Heartbeat: no tasks (HEARTBEAT.md empty)")
|
||||
return
|
||||
|
||||
logger.info("Heartbeat: checking for tasks...")
|
||||
logger.info("Heartbeat: user idle, checking for tasks...")
|
||||
|
||||
try:
|
||||
action, tasks = await self._decide(content)
|
||||
if self.on_heartbeat:
|
||||
try:
|
||||
# Call with suppress_output metadata
|
||||
await self.on_heartbeat(
|
||||
HEARTBEAT_PROMPT,
|
||||
metadata={"suppress_output": True}
|
||||
)
|
||||
|
||||
if action != "run":
|
||||
logger.info("Heartbeat: OK (nothing to report)")
|
||||
return
|
||||
# Note: HEARTBEAT_OK check removed - suppress mode makes it unnecessary
|
||||
logger.info("Heartbeat: completed")
|
||||
|
||||
logger.info("Heartbeat: tasks found, executing...")
|
||||
if self.on_execute:
|
||||
response = await self.on_execute(tasks)
|
||||
if response and self.on_notify:
|
||||
logger.info("Heartbeat: completed, delivering response")
|
||||
await self.on_notify(response)
|
||||
except Exception:
|
||||
logger.exception("Heartbeat execution failed")
|
||||
except Exception as e:
|
||||
logger.error(f"Heartbeat execution failed: {e}")
|
||||
|
||||
async def trigger_now(self) -> str | None:
|
||||
"""Manually trigger a heartbeat."""
|
||||
content = self._read_heartbeat_file()
|
||||
if not content:
|
||||
return None
|
||||
action, tasks = await self._decide(content)
|
||||
if action != "run" or not self.on_execute:
|
||||
return None
|
||||
return await self.on_execute(tasks)
|
||||
if self.on_heartbeat:
|
||||
return await self.on_heartbeat(HEARTBEAT_PROMPT, metadata={"suppress_output": True})
|
||||
return None
|
||||
|
||||
@@ -0,0 +1,139 @@
|
||||
"""HTTP hooks server for external service integration."""
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import uuid
|
||||
|
||||
from aiohttp import web
|
||||
from loguru import logger
|
||||
|
||||
from nanobot.bus.events import InboundMessage
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.config.schema import HooksConfig
|
||||
|
||||
|
||||
class HooksServer:
|
||||
"""
|
||||
HTTP server exposing a /hooks endpoint.
|
||||
|
||||
External services POST JSON messages. The server publishes them
|
||||
to the bus as InboundMessages and uses bus-level correlation
|
||||
to return the agent's response synchronously.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
host: str,
|
||||
port: int,
|
||||
config: HooksConfig,
|
||||
bus: MessageBus,
|
||||
):
|
||||
self.host = host
|
||||
self.port = port
|
||||
self.config = config
|
||||
self.bus = bus
|
||||
self._app = web.Application()
|
||||
self._app.router.add_post(self.config.path, self._handle_hook)
|
||||
self._app.router.add_get("/health", self._handle_health)
|
||||
self._runner: web.AppRunner | None = None
|
||||
|
||||
async def start(self) -> None:
|
||||
"""Start the HTTP server."""
|
||||
if not self.config.has_tokens:
|
||||
logger.warning("Hooks server has no tokens configured — endpoint disabled for security")
|
||||
return
|
||||
|
||||
self._runner = web.AppRunner(self._app)
|
||||
await self._runner.setup()
|
||||
site = web.TCPSite(self._runner, self.host, self.port)
|
||||
await site.start()
|
||||
logger.info(f"Hooks server listening on {self.host}:{self.port}{self.config.path}")
|
||||
|
||||
async def stop(self) -> None:
|
||||
"""Stop the HTTP server."""
|
||||
if self._runner:
|
||||
await self._runner.cleanup()
|
||||
self._runner = None
|
||||
|
||||
def _resolve_auth(self, request: web.Request) -> str | None:
|
||||
"""
|
||||
Validate auth and return token name if valid, None otherwise.
|
||||
Checks Authorization: Bearer <token> and X-Hook-Token headers.
|
||||
"""
|
||||
# Try Authorization: Bearer <token>
|
||||
auth = request.headers.get("Authorization", "")
|
||||
if auth.startswith("Bearer "):
|
||||
token = auth[7:]
|
||||
else:
|
||||
# Try X-Hook-Token header
|
||||
token = request.headers.get("X-Hook-Token", "")
|
||||
|
||||
return self.config.resolve_token(token) if token else None
|
||||
|
||||
async def _handle_health(self, request: web.Request) -> web.Response:
|
||||
"""Health check endpoint — no auth required."""
|
||||
return web.json_response({"status": "ok"})
|
||||
|
||||
async def _handle_hook(self, request: web.Request) -> web.Response:
|
||||
"""Handle incoming hook request."""
|
||||
# Auth check — resolve token name
|
||||
token_name = self._resolve_auth(request)
|
||||
if not token_name:
|
||||
return web.json_response({"error": "unauthorized"}, status=401)
|
||||
|
||||
# Parse body
|
||||
try:
|
||||
body = await request.json()
|
||||
except (json.JSONDecodeError, Exception):
|
||||
return web.json_response({"error": "invalid JSON body"}, status=400)
|
||||
|
||||
# Validate required fields
|
||||
message = body.get("message")
|
||||
if not message or not isinstance(message, str):
|
||||
return web.json_response(
|
||||
{"error": "missing or invalid 'message' field"}, status=400
|
||||
)
|
||||
|
||||
# Optional fields
|
||||
channel = body.get("channel", "hook")
|
||||
chat_id = body.get("chat_id", token_name)
|
||||
timeout = body.get("timeout", self.config.timeout_seconds)
|
||||
|
||||
# Create correlation
|
||||
correlation_id = str(uuid.uuid4())
|
||||
|
||||
# Build InboundMessage
|
||||
msg = InboundMessage(
|
||||
channel=channel,
|
||||
sender_id=f"hook:{token_name}",
|
||||
chat_id=str(chat_id),
|
||||
content=message,
|
||||
metadata={
|
||||
"correlation_id": correlation_id,
|
||||
"hook_source": token_name,
|
||||
},
|
||||
)
|
||||
|
||||
# Fire-and-forget mode
|
||||
if timeout == 0:
|
||||
await self.bus.publish_inbound(msg)
|
||||
return web.json_response({"ok": True}, status=202)
|
||||
|
||||
# Request-response mode
|
||||
future = self.bus.register_correlation(correlation_id)
|
||||
await self.bus.publish_inbound(msg)
|
||||
|
||||
try:
|
||||
response = await asyncio.wait_for(future, timeout=timeout)
|
||||
return web.json_response({"ok": True, "response": response})
|
||||
except asyncio.TimeoutError:
|
||||
return web.json_response(
|
||||
{"ok": False, "error": f"agent did not respond within {timeout}s"},
|
||||
status=504,
|
||||
)
|
||||
except Exception as e:
|
||||
logger.error(f"Hook processing error: {e}")
|
||||
return web.json_response({"error": "internal error"}, status=500)
|
||||
finally:
|
||||
# Clean up correlation on any failure
|
||||
self.bus.cancel_correlation(correlation_id)
|
||||
@@ -1,7 +1,47 @@
|
||||
"""LLM provider abstraction module."""
|
||||
"""Provider module exports."""
|
||||
|
||||
from nanobot.providers.base import LLMProvider, LLMResponse
|
||||
from nanobot.providers.base import LLMProvider, LLMResponse, LongContextError, ToolCallRequest
|
||||
from nanobot.providers.litellm_provider import LiteLLMProvider
|
||||
from nanobot.providers.openai_codex_provider import OpenAICodexProvider
|
||||
from nanobot.providers.anthropic_oauth import AnthropicOAuthProvider
|
||||
from nanobot.providers.registry import should_use_oauth_provider
|
||||
|
||||
__all__ = ["LLMProvider", "LLMResponse", "LiteLLMProvider", "OpenAICodexProvider"]
|
||||
__all__ = [
|
||||
"LLMProvider",
|
||||
"LLMResponse",
|
||||
"ToolCallRequest",
|
||||
"LiteLLMProvider",
|
||||
"OpenAICodexProvider",
|
||||
"AnthropicOAuthProvider",
|
||||
"create_provider",
|
||||
]
|
||||
|
||||
|
||||
def create_provider(
|
||||
api_key: str,
|
||||
model: str,
|
||||
api_base: str | None = None,
|
||||
extra_headers: dict[str, str] | None = None,
|
||||
provider_name: str | None = None,
|
||||
thinking_budget: int = 0,
|
||||
) -> LLMProvider:
|
||||
"""Factory function to create appropriate provider.
|
||||
|
||||
Automatically selects AnthropicOAuthProvider for OAuth tokens,
|
||||
LiteLLMProvider for everything else.
|
||||
"""
|
||||
if should_use_oauth_provider(api_key, model):
|
||||
return AnthropicOAuthProvider(
|
||||
oauth_token=api_key,
|
||||
default_model=model,
|
||||
api_base=api_base,
|
||||
thinking_budget=thinking_budget,
|
||||
)
|
||||
|
||||
return LiteLLMProvider(
|
||||
api_key=api_key,
|
||||
api_base=api_base,
|
||||
default_model=model,
|
||||
extra_headers=extra_headers,
|
||||
provider_name=provider_name,
|
||||
)
|
||||
|
||||
@@ -0,0 +1,680 @@
|
||||
"""Anthropic OAuth provider - direct API calls with Bearer auth.
|
||||
|
||||
This provider bypasses litellm to properly handle OAuth tokens
|
||||
which require Authorization: Bearer header instead of x-api-key.
|
||||
"""
|
||||
|
||||
import json
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
from loguru import logger
|
||||
|
||||
from nanobot.providers.base import LLMProvider, LLMResponse, LongContextError, ToolCallRequest
|
||||
from nanobot.providers.oauth_utils import get_auth_headers, get_claude_code_system_prefix
|
||||
|
||||
|
||||
class AnthropicOAuthProvider(LLMProvider):
|
||||
"""
|
||||
Anthropic provider using OAuth token authentication.
|
||||
|
||||
Unlike the LiteLLM provider, this calls the Anthropic API directly
|
||||
with proper Bearer token authentication for Claude Max/Pro subscriptions.
|
||||
"""
|
||||
|
||||
ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages"
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
oauth_token: str,
|
||||
default_model: str = "claude-opus-4-7",
|
||||
api_base: str | None = None,
|
||||
thinking_budget: int = 0,
|
||||
):
|
||||
super().__init__(api_key=None, api_base=api_base)
|
||||
self.oauth_token = oauth_token
|
||||
self.default_model = default_model
|
||||
self.thinking_budget = thinking_budget
|
||||
self._client: httpx.AsyncClient | None = None
|
||||
|
||||
def _get_headers(self) -> dict[str, str]:
|
||||
"""Get request headers with Bearer auth."""
|
||||
return get_auth_headers(self.oauth_token, is_oauth=True)
|
||||
|
||||
def _get_api_url(self) -> str:
|
||||
"""Get API endpoint URL."""
|
||||
if self.api_base:
|
||||
return f"{self.api_base.rstrip('/')}/v1/messages"
|
||||
return self.ANTHROPIC_API_URL
|
||||
|
||||
@staticmethod
|
||||
def _normalize_model(model: str) -> str:
|
||||
"""Normalize model name for the Anthropic API.
|
||||
|
||||
Anthropic model IDs use hyphens (claude-sonnet-4-6), but users often
|
||||
write dots (claude-sonnet-4.6). Normalize so both work.
|
||||
"""
|
||||
return model.replace(".", "-")
|
||||
|
||||
async def _get_client(self) -> httpx.AsyncClient:
|
||||
"""Get or create async HTTP client."""
|
||||
if self._client is None:
|
||||
self._client = httpx.AsyncClient(
|
||||
timeout=httpx.Timeout(300.0, pool=30.0),
|
||||
)
|
||||
return self._client
|
||||
|
||||
async def _reset_client(self) -> None:
|
||||
"""Destroy and recreate the HTTP client after connection errors."""
|
||||
old = self._client
|
||||
self._client = None
|
||||
if old:
|
||||
try:
|
||||
await old.aclose()
|
||||
except Exception:
|
||||
pass
|
||||
logger.warning("Reset httpx client (pool recycled)")
|
||||
|
||||
async def _diagnose_connectivity(self) -> None:
|
||||
"""Run diagnostics when ConnectTimeout occurs to understand why."""
|
||||
import socket
|
||||
import asyncio
|
||||
|
||||
# 1. Raw socket test (bypasses httpx entirely)
|
||||
try:
|
||||
t0 = __import__('time').monotonic()
|
||||
s = socket.create_connection(('api.anthropic.com', 443), timeout=10)
|
||||
elapsed = __import__('time').monotonic() - t0
|
||||
s.close()
|
||||
logger.warning(f"DIAG: raw socket connect OK in {elapsed:.3f}s")
|
||||
except Exception as e:
|
||||
logger.error(f"DIAG: raw socket connect FAILED: {e}")
|
||||
|
||||
# 2. asyncio connect test (same event loop)
|
||||
try:
|
||||
t0 = __import__('time').monotonic()
|
||||
reader, writer = await asyncio.wait_for(
|
||||
asyncio.open_connection('api.anthropic.com', 443),
|
||||
timeout=10.0,
|
||||
)
|
||||
elapsed = __import__('time').monotonic() - t0
|
||||
writer.close()
|
||||
await writer.wait_closed()
|
||||
logger.warning(f"DIAG: asyncio connect OK in {elapsed:.3f}s")
|
||||
except Exception as e:
|
||||
logger.error(f"DIAG: asyncio connect FAILED: {e}")
|
||||
|
||||
# 3. Fresh httpx client test (new pool)
|
||||
try:
|
||||
t0 = __import__('time').monotonic()
|
||||
async with httpx.AsyncClient(timeout=10.0) as fresh:
|
||||
r = await fresh.get('https://api.anthropic.com/')
|
||||
elapsed = __import__('time').monotonic() - t0
|
||||
logger.warning(f"DIAG: fresh httpx OK in {elapsed:.3f}s (status={r.status_code})")
|
||||
except Exception as e:
|
||||
logger.error(f"DIAG: fresh httpx FAILED: {e}")
|
||||
|
||||
# 4. DNS resolution
|
||||
try:
|
||||
ips = socket.getaddrinfo('api.anthropic.com', 443)
|
||||
logger.warning(f"DIAG: DNS resolved to {len(ips)} entries, first={ips[0][4][0]}")
|
||||
except Exception as e:
|
||||
logger.error(f"DIAG: DNS FAILED: {e}")
|
||||
|
||||
# 5. Connection pool state of the broken client
|
||||
if self._client:
|
||||
transport = self._client._transport
|
||||
if hasattr(transport, '_pool'):
|
||||
pool = transport._pool
|
||||
conns = getattr(pool, '_connections', [])
|
||||
reqs = getattr(pool, '_requests', [])
|
||||
logger.warning(
|
||||
f"DIAG: pool state: {len(conns)} connections, "
|
||||
f"{len(reqs)} pending requests"
|
||||
)
|
||||
for i, conn in enumerate(conns[:5]):
|
||||
state = getattr(conn, '_state', 'unknown')
|
||||
logger.warning(f"DIAG: conn[{i}] state={state}")
|
||||
|
||||
def _prepare_messages(
|
||||
self,
|
||||
messages: list[dict[str, Any]]
|
||||
) -> tuple[str | None, list[dict[str, Any]]]:
|
||||
"""Prepare messages: extract system prompt and convert OpenAI format to Anthropic.
|
||||
|
||||
The agent loop produces messages in OpenAI format:
|
||||
- assistant msgs with tool_calls [{type:"function", function:{name, arguments}}]
|
||||
- tool role msgs with tool_call_id, name, content
|
||||
|
||||
Anthropic API expects:
|
||||
- assistant msgs with content blocks [{type:"tool_use", id, name, input}]
|
||||
- user msgs with content blocks [{type:"tool_result", tool_use_id, content}]
|
||||
|
||||
Returns (system_prompt, anthropic_messages)
|
||||
"""
|
||||
system_parts = []
|
||||
converted: list[dict[str, Any]] = []
|
||||
|
||||
for msg in messages:
|
||||
role = msg.get("role")
|
||||
|
||||
if role == "system":
|
||||
system_parts.append(msg.get("content", ""))
|
||||
continue
|
||||
|
||||
if role == "assistant" and msg.get("tool_calls"):
|
||||
# Convert OpenAI tool_calls to Anthropic content blocks
|
||||
content_blocks: list[dict[str, Any]] = []
|
||||
# Preserve thinking blocks (list=raw API blocks with signatures, str=legacy)
|
||||
rc = msg.get("reasoning_content")
|
||||
if isinstance(rc, list):
|
||||
content_blocks.extend(rc)
|
||||
elif isinstance(rc, str) and rc:
|
||||
content_blocks.append({"type": "thinking", "thinking": rc})
|
||||
text = msg.get("content")
|
||||
if text:
|
||||
content_blocks.append({"type": "text", "text": text})
|
||||
for tc in msg["tool_calls"]:
|
||||
func = tc.get("function", {})
|
||||
args = func.get("arguments", "{}")
|
||||
if isinstance(args, str):
|
||||
try:
|
||||
args = json.loads(args)
|
||||
except (json.JSONDecodeError, TypeError):
|
||||
args = {}
|
||||
content_blocks.append({
|
||||
"type": "tool_use",
|
||||
"id": tc.get("id", ""),
|
||||
"name": func.get("name", ""),
|
||||
"input": args,
|
||||
})
|
||||
converted.append({"role": "assistant", "content": content_blocks})
|
||||
continue
|
||||
|
||||
if role == "assistant" and msg.get("reasoning_content"):
|
||||
# Plain assistant message with thinking (no tool calls)
|
||||
rc = msg["reasoning_content"]
|
||||
if isinstance(rc, list):
|
||||
content_blocks = list(rc)
|
||||
else:
|
||||
content_blocks = [{"type": "thinking", "thinking": rc}]
|
||||
text = msg.get("content")
|
||||
if text:
|
||||
content_blocks.append({"type": "text", "text": text})
|
||||
converted.append({"role": "assistant", "content": content_blocks})
|
||||
continue
|
||||
|
||||
if role == "tool":
|
||||
# Convert tool result to Anthropic user message with tool_result block
|
||||
tool_result_block = {
|
||||
"type": "tool_result",
|
||||
"tool_use_id": msg.get("tool_call_id", ""),
|
||||
"content": msg.get("content", ""),
|
||||
}
|
||||
# Merge into previous user message if it already has tool_result blocks
|
||||
if converted and converted[-1].get("role") == "user":
|
||||
prev_content = converted[-1].get("content")
|
||||
if isinstance(prev_content, list):
|
||||
prev_content.append(tool_result_block)
|
||||
continue
|
||||
converted.append({"role": "user", "content": [tool_result_block]})
|
||||
continue
|
||||
|
||||
if role == "user":
|
||||
content = msg.get("content", "")
|
||||
# Convert OpenAI image_url blocks to Anthropic image blocks
|
||||
if isinstance(content, list):
|
||||
content = self._convert_image_blocks(content)
|
||||
# Merge text into previous user message if it has tool_result blocks
|
||||
# (handles the "Reflect on the results" interleaved message)
|
||||
if converted and converted[-1].get("role") == "user":
|
||||
prev_content = converted[-1].get("content")
|
||||
if isinstance(prev_content, list):
|
||||
if isinstance(content, str):
|
||||
prev_content.append({"type": "text", "text": content})
|
||||
elif isinstance(content, list):
|
||||
prev_content.extend(content)
|
||||
continue
|
||||
converted.append({"role": role, "content": content})
|
||||
continue
|
||||
|
||||
# Pass through other messages (assistant without tool_calls, etc.)
|
||||
converted.append(msg)
|
||||
|
||||
system_prompt = "\n\n".join(system_parts)
|
||||
return system_prompt, converted
|
||||
|
||||
@staticmethod
|
||||
def _convert_image_blocks(content: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
"""Convert OpenAI image_url blocks to Anthropic image blocks.
|
||||
|
||||
OpenAI format: {"type": "image_url", "image_url": {"url": "data:mime;base64,DATA"}}
|
||||
Anthropic format: {"type": "image", "source": {"type": "base64", "media_type": "mime", "data": "DATA"}}
|
||||
"""
|
||||
converted = []
|
||||
for block in content:
|
||||
if block.get("type") == "image_url":
|
||||
url = block.get("image_url", {}).get("url", "")
|
||||
if url.startswith("data:") and ";base64," in url:
|
||||
header, data = url.split(";base64,", 1)
|
||||
media_type = header.removeprefix("data:")
|
||||
converted.append({
|
||||
"type": "image",
|
||||
"source": {"type": "base64", "media_type": media_type, "data": data},
|
||||
})
|
||||
else:
|
||||
converted.append({
|
||||
"type": "image",
|
||||
"source": {"type": "url", "url": url},
|
||||
})
|
||||
else:
|
||||
converted.append(block)
|
||||
return converted
|
||||
|
||||
def _convert_tools_to_anthropic(
|
||||
self,
|
||||
tools: list[dict[str, Any]] | list[Any] | None
|
||||
) -> list[dict[str, Any]] | None:
|
||||
"""Convert tools to Anthropic API format.
|
||||
|
||||
Supports both function tools (custom) and native tools (Anthropic).
|
||||
Function tools are converted to Anthropic format.
|
||||
Native tools are passed through unchanged.
|
||||
Tool objects (with to_params/to_schema methods) are converted to dicts.
|
||||
"""
|
||||
if not tools:
|
||||
return None
|
||||
|
||||
anthropic_tools = []
|
||||
for tool in tools:
|
||||
# Convert tool objects to dicts first
|
||||
if hasattr(tool, 'to_params'): # Native Anthropic tool
|
||||
tool_dict = tool.to_params()
|
||||
elif hasattr(tool, 'to_schema'): # Function tool
|
||||
tool_dict = tool.to_schema()
|
||||
else:
|
||||
tool_dict = tool # Already a dict
|
||||
|
||||
# Now process the dict
|
||||
if tool_dict.get("type") == "function":
|
||||
# Convert function tool format
|
||||
func = tool_dict["function"]
|
||||
anthropic_tools.append({
|
||||
"name": func["name"],
|
||||
"description": func.get("description", ""),
|
||||
"input_schema": func.get("parameters", {"type": "object", "properties": {}})
|
||||
})
|
||||
else:
|
||||
# Pass through native tool format as-is
|
||||
# (bash_20250124, text_editor_20250728, computer_20251124, etc.)
|
||||
anthropic_tools.append(tool_dict)
|
||||
|
||||
return anthropic_tools if anthropic_tools else None
|
||||
|
||||
async def _make_request(
|
||||
self,
|
||||
messages: list[dict[str, Any]],
|
||||
system: str | None = None,
|
||||
model: str = "claude-opus-4-5",
|
||||
max_tokens: int = 4096,
|
||||
temperature: float = 0.7,
|
||||
tools: list[dict[str, Any]] | None = None,
|
||||
thinking_budget_override: int | None = None,
|
||||
context_management: dict[str, Any] | None = None,
|
||||
beta_flags: set[str] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Make request to Anthropic API."""
|
||||
client = await self._get_client()
|
||||
|
||||
# Add cache breakpoints on the last TWO user messages (4-breakpoint strategy):
|
||||
# BP3: Second-to-last user message (stable history from previous turn)
|
||||
# BP4: Last user message (current turn, will become BP3 next turn)
|
||||
# This allows BP3 to reuse what BP4 cached last turn.
|
||||
user_indices = [i for i, m in enumerate(messages) if m.get("role") == "user"]
|
||||
|
||||
if len(user_indices) >= 2:
|
||||
# BP3: Second-to-last user message
|
||||
idx = user_indices[-2]
|
||||
msg = messages[idx]
|
||||
content = msg["content"]
|
||||
if isinstance(content, str):
|
||||
messages[idx] = {**msg, "content": [{"type": "text", "text": content, "cache_control": {"type": "ephemeral"}}]}
|
||||
elif isinstance(content, list) and content:
|
||||
new_content = list(content)
|
||||
new_content[-1] = {**new_content[-1], "cache_control": {"type": "ephemeral"}}
|
||||
messages[idx] = {**msg, "content": new_content}
|
||||
|
||||
if len(user_indices) >= 1:
|
||||
# BP4: Last user message
|
||||
idx = user_indices[-1]
|
||||
msg = messages[idx]
|
||||
content = msg["content"]
|
||||
if isinstance(content, str):
|
||||
messages[idx] = {**msg, "content": [{"type": "text", "text": content, "cache_control": {"type": "ephemeral"}}]}
|
||||
elif isinstance(content, list) and content:
|
||||
new_content = list(content)
|
||||
new_content[-1] = {**new_content[-1], "cache_control": {"type": "ephemeral"}}
|
||||
messages[idx] = {**msg, "content": new_content}
|
||||
|
||||
payload: dict[str, Any] = {
|
||||
"model": model,
|
||||
"messages": messages,
|
||||
"max_tokens": max_tokens,
|
||||
}
|
||||
|
||||
# Extended thinking: temperature must be 1 when enabled
|
||||
effective_thinking = thinking_budget_override if thinking_budget_override is not None else self.thinking_budget
|
||||
if effective_thinking > 0:
|
||||
payload["temperature"] = 1
|
||||
# max_tokens must exceed budget_tokens
|
||||
if max_tokens <= effective_thinking:
|
||||
payload["max_tokens"] = effective_thinking + 4096
|
||||
payload["thinking"] = {
|
||||
"type": "enabled",
|
||||
"budget_tokens": effective_thinking,
|
||||
}
|
||||
else:
|
||||
payload["temperature"] = temperature
|
||||
|
||||
if system:
|
||||
payload["system"] = [
|
||||
{"type": "text", "text": get_claude_code_system_prefix()},
|
||||
{"type": "text", "text": system, "cache_control": {"type": "ephemeral", "ttl": "1h"}},
|
||||
]
|
||||
else:
|
||||
payload["system"] = [
|
||||
{"type": "text", "text": get_claude_code_system_prefix()},
|
||||
]
|
||||
|
||||
if tools:
|
||||
cached_tools = list(tools)
|
||||
cached_tools[-1] = {**cached_tools[-1], "cache_control": {"type": "ephemeral", "ttl": "1h"}}
|
||||
payload["tools"] = cached_tools
|
||||
|
||||
if context_management:
|
||||
payload["context_management"] = context_management
|
||||
|
||||
edit_types = [e.get("type") for e in (context_management or {}).get("edits", [])]
|
||||
|
||||
# Build headers with beta flags if provided
|
||||
headers = self._get_headers()
|
||||
if beta_flags:
|
||||
# Merge with existing beta header (from OAuth hardcoded flags)
|
||||
existing_beta = headers.get("anthropic-beta", "")
|
||||
existing_flags = set(existing_beta.split(",")) if existing_beta else set()
|
||||
all_flags = existing_flags | beta_flags
|
||||
headers["anthropic-beta"] = ",".join(sorted(all_flags))
|
||||
|
||||
logger.info(
|
||||
"Anthropic request: model={} max_tokens={} thinking={} tools={} context_mgmt={} beta={}",
|
||||
payload.get("model"), payload.get("max_tokens"),
|
||||
payload.get("thinking", "disabled"),
|
||||
len(payload.get("tools", [])),
|
||||
edit_types or "none",
|
||||
headers.get("anthropic-beta", "none"),
|
||||
)
|
||||
|
||||
# Debug: Log tool names for diagnostic purposes
|
||||
if payload.get("tools"):
|
||||
tool_names = [t.get("name", "unnamed") for t in payload["tools"]]
|
||||
logger.debug(f"Tool names in request: {tool_names}")
|
||||
|
||||
# Debug: Log message structure to diagnose orphaned tool_result errors
|
||||
for idx, m in enumerate(payload.get("messages", [])):
|
||||
role = m.get("role", "?")
|
||||
content = m.get("content", "")
|
||||
if isinstance(content, list):
|
||||
block_types = [b.get("type", "?") for b in content]
|
||||
logger.debug(f" msg[{idx}] role={role} blocks={block_types}")
|
||||
else:
|
||||
logger.debug(f" msg[{idx}] role={role} text={str(content)[:80]}")
|
||||
|
||||
import asyncio
|
||||
import time as _time
|
||||
|
||||
max_retries = 3
|
||||
base_delay = 2.0 # seconds
|
||||
|
||||
for attempt in range(max_retries + 1):
|
||||
_t0 = _time.monotonic()
|
||||
try:
|
||||
response = await client.post(
|
||||
self._get_api_url(),
|
||||
headers=headers,
|
||||
json=payload,
|
||||
)
|
||||
except httpx.ConnectTimeout:
|
||||
elapsed = _time.monotonic() - _t0
|
||||
logger.error(f"ConnectTimeout after {elapsed:.1f}s (attempt {attempt+1}/{max_retries+1})")
|
||||
if attempt == 0:
|
||||
await self._diagnose_connectivity()
|
||||
await self._reset_client()
|
||||
if attempt < max_retries:
|
||||
delay = base_delay * (2 ** attempt)
|
||||
logger.info(f"Retrying in {delay:.1f}s...")
|
||||
await asyncio.sleep(delay)
|
||||
continue
|
||||
raise
|
||||
except httpx.PoolTimeout:
|
||||
elapsed = _time.monotonic() - _t0
|
||||
logger.error(f"PoolTimeout after {elapsed:.1f}s (attempt {attempt+1}/{max_retries+1})")
|
||||
await self._reset_client()
|
||||
if attempt < max_retries:
|
||||
delay = base_delay * (2 ** attempt)
|
||||
logger.info(f"Retrying in {delay:.1f}s...")
|
||||
await asyncio.sleep(delay)
|
||||
continue
|
||||
raise
|
||||
except (httpx.ConnectError, httpx.TimeoutException) as e:
|
||||
elapsed = _time.monotonic() - _t0
|
||||
logger.error(f"{type(e).__name__} after {elapsed:.1f}s (attempt {attempt+1}/{max_retries+1})")
|
||||
if attempt < max_retries:
|
||||
delay = base_delay * (2 ** attempt)
|
||||
logger.info(f"Retrying in {delay:.1f}s...")
|
||||
await asyncio.sleep(delay)
|
||||
continue
|
||||
raise
|
||||
elapsed = _time.monotonic() - _t0
|
||||
if elapsed > 30:
|
||||
logger.warning(f"Anthropic API slow response: {elapsed:.1f}s")
|
||||
|
||||
# Dump rate limit headers for analysis
|
||||
try:
|
||||
import datetime
|
||||
import os
|
||||
header_dump = {
|
||||
"timestamp": datetime.datetime.now(datetime.UTC).isoformat(),
|
||||
"status_code": response.status_code,
|
||||
"model": payload.get("model"),
|
||||
"headers": dict(response.headers),
|
||||
}
|
||||
dump_path = "/root/.nanobot/workspace/api_headers.jsonl"
|
||||
with open(dump_path, "a") as f:
|
||||
f.write(json.dumps(header_dump) + "\n")
|
||||
# Capture rate limit state for quota-based model switching
|
||||
hdrs = response.headers
|
||||
rate_limit_state = {
|
||||
"updated_at": datetime.datetime.utcnow().isoformat(),
|
||||
"model": payload.get("model"),
|
||||
"weekly_all_models": float(hdrs["anthropic-ratelimit-unified-7d-utilization"]) if hdrs.get("anthropic-ratelimit-unified-7d-utilization") else None,
|
||||
"weekly_sonnet": float(hdrs["anthropic-ratelimit-unified-7d_sonnet-utilization"]) if hdrs.get("anthropic-ratelimit-unified-7d_sonnet-utilization") else None,
|
||||
"session_5h": float(hdrs["anthropic-ratelimit-unified-5h-utilization"]) if hdrs.get("anthropic-ratelimit-unified-5h-utilization") else None,
|
||||
"weekly_reset": int(hdrs["anthropic-ratelimit-unified-7d-reset"]) if hdrs.get("anthropic-ratelimit-unified-7d-reset") else None,
|
||||
"session_reset": int(hdrs["anthropic-ratelimit-unified-5h-reset"]) if hdrs.get("anthropic-ratelimit-unified-5h-reset") else None,
|
||||
"binding_limit": hdrs.get("anthropic-ratelimit-unified-representative-claim"),
|
||||
"sonnet_fallback": hdrs.get("anthropic-ratelimit-unified-fallback"),
|
||||
}
|
||||
state_path = "/root/.nanobot/workspace/memory/rate_limits.json"
|
||||
os.makedirs(os.path.dirname(state_path), exist_ok=True)
|
||||
with open(state_path, "w") as f:
|
||||
json.dump(rate_limit_state, f, indent=2)
|
||||
except Exception as e:
|
||||
logger.warning("Rate limit header capture failed: {}", e)
|
||||
|
||||
# Retry on 5xx server errors and 429 rate limits
|
||||
if response.status_code >= 500 or response.status_code == 429:
|
||||
error_text = response.text
|
||||
logger.warning(f"Anthropic API {response.status_code} (attempt {attempt+1}/{max_retries+1}): {error_text[:200]}")
|
||||
|
||||
# Long context 429 — retrying won't help, need to trim context
|
||||
if response.status_code == 429 and "long context" in error_text.lower():
|
||||
raise LongContextError(f"Context too long for current plan: {error_text[:200]}")
|
||||
|
||||
if attempt < max_retries:
|
||||
if response.status_code == 429:
|
||||
retry_after = response.headers.get("retry-after")
|
||||
delay = float(retry_after) if retry_after else base_delay * (2 ** attempt)
|
||||
else:
|
||||
delay = base_delay * (2 ** attempt)
|
||||
logger.info(f"Retrying in {delay:.1f}s...")
|
||||
await asyncio.sleep(delay)
|
||||
continue
|
||||
raise Exception(f"Anthropic API error {response.status_code}: {error_text}")
|
||||
|
||||
if response.status_code != 200:
|
||||
error_text = response.text
|
||||
raise Exception(f"Anthropic API error {response.status_code}: {error_text}")
|
||||
|
||||
return response.json()
|
||||
|
||||
# Should not reach here, but just in case
|
||||
raise Exception("Exhausted all retry attempts")
|
||||
|
||||
async def chat(
|
||||
self,
|
||||
messages: list[dict[str, Any]],
|
||||
tools: list[dict[str, Any]] | list[Any] | None = None,
|
||||
model: str | None = None,
|
||||
max_tokens: int = 4096,
|
||||
temperature: float = 0.7,
|
||||
thinking_budget: int | None = None,
|
||||
context_management: dict[str, Any] | None = None,
|
||||
) -> LLMResponse:
|
||||
"""Send chat completion request to Anthropic API."""
|
||||
model = model or self.default_model
|
||||
|
||||
# Strip provider prefix if present (e.g. "anthropic/claude-opus-4-5" -> "claude-opus-4-5")
|
||||
if "/" in model:
|
||||
model = model.split("/")[-1]
|
||||
|
||||
# Normalize dots to hyphens (claude-sonnet-4.6 -> claude-sonnet-4-6)
|
||||
model = self._normalize_model(model)
|
||||
|
||||
system, prepared_messages = self._prepare_messages(messages)
|
||||
|
||||
# Collect beta flags from native tools BEFORE conversion
|
||||
beta_flags: set[str] = set()
|
||||
if tools:
|
||||
for tool in tools:
|
||||
if hasattr(tool, 'beta_flag') and tool.beta_flag:
|
||||
beta_flags.add(tool.beta_flag)
|
||||
|
||||
logger.debug(f"Beta flags collected: {beta_flags} (from {len(tools) if tools else 0} tools)")
|
||||
|
||||
# Convert tools to API format
|
||||
anthropic_tools = self._convert_tools_to_anthropic(tools)
|
||||
|
||||
# Per-call thinking override (None = use instance default)
|
||||
effective_thinking = self.thinking_budget if thinking_budget is None else thinking_budget
|
||||
|
||||
try:
|
||||
response = await self._make_request(
|
||||
messages=prepared_messages,
|
||||
system=system,
|
||||
model=model,
|
||||
max_tokens=max_tokens,
|
||||
temperature=temperature,
|
||||
tools=anthropic_tools,
|
||||
thinking_budget_override=effective_thinking,
|
||||
context_management=context_management,
|
||||
beta_flags=beta_flags,
|
||||
)
|
||||
return self._parse_response(response)
|
||||
except LongContextError:
|
||||
raise # Let caller handle context trimming
|
||||
except Exception as e:
|
||||
logger.exception("Exception in chat():")
|
||||
error_msg = f"{type(e).__name__}: {str(e)}" if str(e) else f"{type(e).__name__} (no message)"
|
||||
return LLMResponse(
|
||||
content=f"Error calling LLM: {error_msg}",
|
||||
finish_reason="error",
|
||||
)
|
||||
|
||||
def _parse_response(self, response: dict[str, Any]) -> LLMResponse:
|
||||
"""Parse Anthropic API response."""
|
||||
content_blocks = response.get("content", [])
|
||||
|
||||
text_content = ""
|
||||
thinking_blocks: list[dict[str, Any]] = []
|
||||
tool_calls = []
|
||||
|
||||
for block in content_blocks:
|
||||
if block.get("type") == "thinking":
|
||||
# Preserve full block including signature for multi-turn replay
|
||||
thinking_blocks.append(block)
|
||||
elif block.get("type") == "text":
|
||||
text_content += block.get("text", "")
|
||||
elif block.get("type") == "tool_use":
|
||||
tool_calls.append(ToolCallRequest(
|
||||
id=block.get("id", ""),
|
||||
name=block.get("name", ""),
|
||||
arguments=block.get("input", {}),
|
||||
))
|
||||
|
||||
usage = {}
|
||||
if "usage" in response:
|
||||
usage = {
|
||||
"prompt_tokens": response["usage"].get("input_tokens", 0),
|
||||
"completion_tokens": response["usage"].get("output_tokens", 0),
|
||||
"total_tokens": (
|
||||
response["usage"].get("input_tokens", 0) +
|
||||
response["usage"].get("output_tokens", 0)
|
||||
),
|
||||
}
|
||||
|
||||
stop_reason = response.get("stop_reason", "end_turn")
|
||||
thinking_chars = sum(len(b.get("thinking", "")) for b in thinking_blocks) if thinking_blocks else 0
|
||||
raw_usage = response.get("usage", {})
|
||||
cache_write = raw_usage.get("cache_creation_input_tokens", 0)
|
||||
cache_read = raw_usage.get("cache_read_input_tokens", 0)
|
||||
logger.info(
|
||||
"Anthropic response: stop={} tool_calls={} thinking={} chars, "
|
||||
"input={} output={} cache_write={} cache_read={} tokens",
|
||||
stop_reason, len(tool_calls), thinking_chars,
|
||||
usage.get("prompt_tokens", 0), usage.get("completion_tokens", 0),
|
||||
cache_write, cache_read,
|
||||
)
|
||||
|
||||
# Log context editing activity if any edits were applied
|
||||
if applied_edits := response.get("context_management", {}).get("applied_edits"):
|
||||
for edit in applied_edits:
|
||||
edit_type = edit.get("type", "?")
|
||||
cleared_tokens = edit.get("cleared_input_tokens", 0)
|
||||
if edit_type == "clear_tool_uses_20250919":
|
||||
logger.info(
|
||||
"Context edit: cleared {} tool uses ({} tokens)",
|
||||
edit.get("cleared_tool_uses", 0), cleared_tokens,
|
||||
)
|
||||
elif edit_type == "clear_thinking_20251015":
|
||||
logger.info(
|
||||
"Context edit: cleared {} thinking turns ({} tokens)",
|
||||
edit.get("cleared_thinking_turns", 0), cleared_tokens,
|
||||
)
|
||||
|
||||
return LLMResponse(
|
||||
content=text_content or None,
|
||||
tool_calls=tool_calls,
|
||||
finish_reason=stop_reason,
|
||||
usage=usage,
|
||||
reasoning_content=thinking_blocks or None,
|
||||
)
|
||||
|
||||
def get_default_model(self) -> str:
|
||||
"""Get the default model."""
|
||||
return self.default_model
|
||||
|
||||
async def close(self):
|
||||
"""Close the HTTP client."""
|
||||
if self._client:
|
||||
await self._client.aclose()
|
||||
self._client = None
|
||||
@@ -20,7 +20,7 @@ class LLMResponse:
|
||||
tool_calls: list[ToolCallRequest] = field(default_factory=list)
|
||||
finish_reason: str = "stop"
|
||||
usage: dict[str, int] = field(default_factory=dict)
|
||||
reasoning_content: str | None = None # Kimi, DeepSeek-R1 etc.
|
||||
reasoning_content: Any = None # str for Kimi/DeepSeek-R1; list[dict] for Anthropic thinking blocks
|
||||
|
||||
@property
|
||||
def has_tool_calls(self) -> bool:
|
||||
@@ -28,6 +28,11 @@ class LLMResponse:
|
||||
return len(self.tool_calls) > 0
|
||||
|
||||
|
||||
class LongContextError(Exception):
|
||||
"""Raised when the API rejects a request due to long context limits."""
|
||||
pass
|
||||
|
||||
|
||||
class LLMProvider(ABC):
|
||||
"""
|
||||
Abstract base class for LLM providers.
|
||||
@@ -88,6 +93,8 @@ class LLMProvider(ABC):
|
||||
model: str | None = None,
|
||||
max_tokens: int = 4096,
|
||||
temperature: float = 0.7,
|
||||
thinking_budget: int | None = None,
|
||||
context_management: dict[str, Any] | None = None,
|
||||
) -> LLMResponse:
|
||||
"""
|
||||
Send a chat completion request.
|
||||
|
||||
@@ -37,7 +37,7 @@ class LiteLLMProvider(LLMProvider):
|
||||
self,
|
||||
api_key: str | None = None,
|
||||
api_base: str | None = None,
|
||||
default_model: str = "anthropic/claude-opus-4-5",
|
||||
default_model: str = "anthropic/claude-opus-4-7",
|
||||
extra_headers: dict[str, str] | None = None,
|
||||
provider_name: str | None = None,
|
||||
):
|
||||
@@ -178,6 +178,8 @@ class LiteLLMProvider(LLMProvider):
|
||||
model: str | None = None,
|
||||
max_tokens: int = 4096,
|
||||
temperature: float = 0.7,
|
||||
thinking_budget: int | None = None,
|
||||
context_management: dict[str, Any] | None = None, # Anthropic-only, ignored here
|
||||
) -> LLMResponse:
|
||||
"""
|
||||
Send a chat completion request via LiteLLM.
|
||||
@@ -185,7 +187,7 @@ class LiteLLMProvider(LLMProvider):
|
||||
Args:
|
||||
messages: List of message dicts with 'role' and 'content'.
|
||||
tools: Optional list of tool definitions in OpenAI format.
|
||||
model: Model identifier (e.g., 'anthropic/claude-sonnet-4-5').
|
||||
model: Model identifier (e.g., 'anthropic/claude-sonnet-4-6').
|
||||
max_tokens: Maximum tokens in response.
|
||||
temperature: Sampling temperature.
|
||||
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
"""OAuth utility functions for Anthropic subscription auth."""
|
||||
|
||||
from typing import Any
|
||||
|
||||
|
||||
def is_oauth_token(token: str | None) -> bool:
|
||||
"""Check if token is an OAuth token (vs regular API key).
|
||||
|
||||
OAuth tokens from Claude Max/Pro contain 'sk-ant-oat' prefix.
|
||||
Regular API keys use 'sk-ant-api03' or similar.
|
||||
"""
|
||||
if not token:
|
||||
return False
|
||||
return "sk-ant-oat" in token
|
||||
|
||||
|
||||
def get_auth_headers(token: str, is_oauth: bool = False) -> dict[str, str]:
|
||||
"""Get authentication headers for Anthropic API.
|
||||
|
||||
OAuth tokens require Authorization: Bearer header.
|
||||
Regular API keys use x-api-key header.
|
||||
"""
|
||||
headers: dict[str, str] = {
|
||||
"anthropic-version": "2023-06-01",
|
||||
"content-type": "application/json",
|
||||
}
|
||||
|
||||
if is_oauth:
|
||||
headers["Authorization"] = f"Bearer {token}"
|
||||
# Required headers to mimic Claude Code client
|
||||
headers["anthropic-beta"] = "claude-code-20250219,oauth-2025-04-20,context-management-2025-06-27"
|
||||
headers["anthropic-dangerous-direct-browser-access"] = "true"
|
||||
headers["user-agent"] = "claude-cli/2.1.2 (external, cli)"
|
||||
headers["x-app"] = "cli"
|
||||
else:
|
||||
headers["x-api-key"] = token
|
||||
|
||||
return headers
|
||||
|
||||
|
||||
def get_claude_code_system_prefix() -> str:
|
||||
"""Get the required system prompt prefix for OAuth tokens.
|
||||
|
||||
Anthropic requires this identity declaration for OAuth auth.
|
||||
"""
|
||||
return "You are a Claude agent, built on Anthropic's Claude Agent SDK."
|
||||
@@ -460,3 +460,24 @@ def find_by_name(name: str) -> ProviderSpec | None:
|
||||
if spec.name == name:
|
||||
return spec
|
||||
return None
|
||||
|
||||
|
||||
def should_use_oauth_provider(api_key: str | None, model: str) -> bool:
|
||||
"""Determine if OAuth provider should be used.
|
||||
|
||||
OAuth provider is used when:
|
||||
1. API key is an OAuth token (contains 'sk-ant-oat')
|
||||
2. Model is an Anthropic model (contains 'claude' or 'anthropic')
|
||||
"""
|
||||
if not api_key:
|
||||
return False
|
||||
|
||||
if "sk-ant-oat" not in api_key:
|
||||
return False
|
||||
|
||||
model_lower = model.lower()
|
||||
anthropic_spec = find_by_name("anthropic")
|
||||
if anthropic_spec:
|
||||
return any(kw in model_lower for kw in anthropic_spec.keywords)
|
||||
|
||||
return False
|
||||
|
||||
+54
-25
@@ -19,9 +19,7 @@ class Session:
|
||||
|
||||
Stores messages in JSONL format for easy reading and persistence.
|
||||
|
||||
Important: Messages are append-only for LLM cache efficiency.
|
||||
The consolidation process writes summaries to MEMORY.md/HISTORY.md
|
||||
but does NOT modify the messages list or get_history() output.
|
||||
Messages are trimmed after consolidation to keep session size manageable.
|
||||
"""
|
||||
|
||||
key: str # channel:chat_id
|
||||
@@ -29,7 +27,6 @@ class Session:
|
||||
created_at: datetime = field(default_factory=datetime.now)
|
||||
updated_at: datetime = field(default_factory=datetime.now)
|
||||
metadata: dict[str, Any] = field(default_factory=dict)
|
||||
last_consolidated: int = 0 # Number of messages already consolidated to files
|
||||
|
||||
def add_message(self, role: str, content: str, **kwargs: Any) -> None:
|
||||
"""Add a message to the session."""
|
||||
@@ -41,31 +38,47 @@ class Session:
|
||||
}
|
||||
self.messages.append(msg)
|
||||
self.updated_at = datetime.now()
|
||||
|
||||
def get_history(self, max_messages: int = 500) -> list[dict[str, Any]]:
|
||||
"""Return unconsolidated messages for LLM input, aligned to a user turn."""
|
||||
unconsolidated = self.messages[self.last_consolidated:]
|
||||
sliced = unconsolidated[-max_messages:]
|
||||
|
||||
# Drop leading non-user messages to avoid orphaned tool_result blocks
|
||||
for i, m in enumerate(sliced):
|
||||
if m.get("role") == "user":
|
||||
sliced = sliced[i:]
|
||||
break
|
||||
def add_raw_message(self, msg: dict[str, Any]) -> None:
|
||||
"""Add a pre-formed message dict to the session, preserving all fields."""
|
||||
stored = dict(msg)
|
||||
if "timestamp" not in stored:
|
||||
stored["timestamp"] = datetime.now().isoformat()
|
||||
self.messages.append(stored)
|
||||
self.updated_at = datetime.now()
|
||||
|
||||
# Fields that are valid in the Anthropic/OpenAI messages API.
|
||||
# Everything else (timestamp, tools_used, etc.) is internal metadata.
|
||||
_API_FIELDS = {"role", "content", "tool_calls", "tool_call_id", "name", "reasoning_content"}
|
||||
|
||||
def get_history(self) -> list[dict[str, Any]]:
|
||||
"""
|
||||
Get full message history for LLM context.
|
||||
|
||||
The server-side context editing API (clear_tool_uses_20250919) handles
|
||||
trimming old tool chains safely at token thresholds, so we send the full
|
||||
history and let the server decide what to drop.
|
||||
|
||||
Messages with ``_hidden_sig`` get a ``[HIDDEN:{sig}]`` prefix applied to
|
||||
their content so the model knows the user never saw them. The prefix is
|
||||
applied at read time (not stored in content) to preserve prompt-cache
|
||||
stability: the same prefixed string is produced every turn.
|
||||
|
||||
Returns:
|
||||
List of messages in LLM format (API-relevant fields only).
|
||||
"""
|
||||
out: list[dict[str, Any]] = []
|
||||
for m in sliced:
|
||||
entry: dict[str, Any] = {"role": m["role"], "content": m.get("content", "")}
|
||||
for k in ("tool_calls", "tool_call_id", "name"):
|
||||
if k in m:
|
||||
entry[k] = m[k]
|
||||
out.append(entry)
|
||||
for m in self.messages:
|
||||
msg = {k: v for k, v in m.items() if k in self._API_FIELDS and v is not None}
|
||||
sig = m.get("_hidden_sig")
|
||||
if sig and isinstance(msg.get("content"), str):
|
||||
msg["content"] = f"[HIDDEN:{sig}] {msg['content']}"
|
||||
out.append(msg)
|
||||
return out
|
||||
|
||||
def clear(self) -> None:
|
||||
"""Clear all messages and reset session to initial state."""
|
||||
self.messages = []
|
||||
self.last_consolidated = 0
|
||||
self.updated_at = datetime.now()
|
||||
|
||||
|
||||
@@ -131,7 +144,6 @@ class SessionManager:
|
||||
messages = []
|
||||
metadata = {}
|
||||
created_at = None
|
||||
last_consolidated = 0
|
||||
|
||||
with open(path, encoding="utf-8") as f:
|
||||
for line in f:
|
||||
@@ -144,7 +156,7 @@ class SessionManager:
|
||||
if data.get("_type") == "metadata":
|
||||
metadata = data.get("metadata", {})
|
||||
created_at = datetime.fromisoformat(data["created_at"]) if data.get("created_at") else None
|
||||
last_consolidated = data.get("last_consolidated", 0)
|
||||
# Ignore legacy last_consolidated field
|
||||
else:
|
||||
messages.append(data)
|
||||
|
||||
@@ -153,7 +165,6 @@ class SessionManager:
|
||||
messages=messages,
|
||||
created_at=created_at or datetime.now(),
|
||||
metadata=metadata,
|
||||
last_consolidated=last_consolidated
|
||||
)
|
||||
except Exception as e:
|
||||
logger.warning("Failed to load session {}: {}", key, e)
|
||||
@@ -170,13 +181,31 @@ class SessionManager:
|
||||
"created_at": session.created_at.isoformat(),
|
||||
"updated_at": session.updated_at.isoformat(),
|
||||
"metadata": session.metadata,
|
||||
"last_consolidated": session.last_consolidated
|
||||
}
|
||||
f.write(json.dumps(metadata_line, ensure_ascii=False) + "\n")
|
||||
for msg in session.messages:
|
||||
f.write(json.dumps(msg, ensure_ascii=False) + "\n")
|
||||
|
||||
self._cache[session.key] = session
|
||||
self._append_audit(session)
|
||||
|
||||
def _append_audit(self, session: Session) -> None:
|
||||
"""Append session state to an audit log (append-only, rotated monthly)."""
|
||||
now = datetime.now()
|
||||
safe_key = safe_filename(session.key.replace(":", "_"))
|
||||
audit_path = self.sessions_dir / f"{safe_key}.audit.{now:%Y-%m}.jsonl"
|
||||
try:
|
||||
with open(audit_path, "a", encoding="utf-8") as f:
|
||||
marker = {
|
||||
"_type": "save_marker",
|
||||
"timestamp": now.isoformat(),
|
||||
"message_count": len(session.messages),
|
||||
}
|
||||
f.write(json.dumps(marker, ensure_ascii=False) + "\n")
|
||||
for msg in session.messages:
|
||||
f.write(json.dumps(msg, ensure_ascii=False) + "\n")
|
||||
except Exception as e:
|
||||
logger.warning("Audit log write failed for {}: {}", session.key, e)
|
||||
|
||||
def invalidate(self, key: str) -> None:
|
||||
"""Remove a session from the in-memory cache."""
|
||||
|
||||
+34
-35
@@ -1,6 +1,6 @@
|
||||
[project]
|
||||
name = "nanobot-ai"
|
||||
version = "0.1.4.post2"
|
||||
version = "0.1.3.post7"
|
||||
description = "A lightweight personal AI assistant framework"
|
||||
requires-python = ">=3.11"
|
||||
license = {text = "MIT"}
|
||||
@@ -17,44 +17,44 @@ classifiers = [
|
||||
]
|
||||
|
||||
dependencies = [
|
||||
"typer>=0.20.0,<1.0.0",
|
||||
"litellm>=1.81.5,<2.0.0",
|
||||
"pydantic>=2.12.0,<3.0.0",
|
||||
"pydantic-settings>=2.12.0,<3.0.0",
|
||||
"websockets>=16.0,<17.0",
|
||||
"websocket-client>=1.9.0,<2.0.0",
|
||||
"httpx>=0.28.0,<1.0.0",
|
||||
"oauth-cli-kit>=0.1.3,<1.0.0",
|
||||
"loguru>=0.7.3,<1.0.0",
|
||||
"readability-lxml>=0.8.4,<1.0.0",
|
||||
"rich>=14.0.0,<15.0.0",
|
||||
"croniter>=6.0.0,<7.0.0",
|
||||
"dingtalk-stream>=0.24.0,<1.0.0",
|
||||
"python-telegram-bot[socks]>=22.0,<23.0",
|
||||
"lark-oapi>=1.5.0,<2.0.0",
|
||||
"socksio>=1.0.0,<2.0.0",
|
||||
"python-socketio>=5.16.0,<6.0.0",
|
||||
"msgpack>=1.1.0,<2.0.0",
|
||||
"slack-sdk>=3.39.0,<4.0.0",
|
||||
"slackify-markdown>=0.2.0,<1.0.0",
|
||||
"qq-botpy>=1.2.0,<2.0.0",
|
||||
"python-socks[asyncio]>=2.8.0,<3.0.0",
|
||||
"prompt-toolkit>=3.0.50,<4.0.0",
|
||||
"mcp>=1.26.0,<2.0.0",
|
||||
"json-repair>=0.57.0,<1.0.0",
|
||||
"typer>=0.9.0",
|
||||
"litellm>=1.0.0",
|
||||
"pydantic>=2.0.0",
|
||||
"pydantic-settings>=2.0.0",
|
||||
"websockets>=12.0",
|
||||
"websocket-client>=1.6.0",
|
||||
"httpx[socks]>=0.25.0",
|
||||
"loguru>=0.7.0",
|
||||
"readability-lxml>=0.8.0",
|
||||
"rich>=13.0.0",
|
||||
"croniter>=2.0.0",
|
||||
"dingtalk-stream>=0.4.0",
|
||||
"python-telegram-bot[socks]>=21.0",
|
||||
"lark-oapi>=1.0.0",
|
||||
"socksio>=1.0.0",
|
||||
"python-socketio>=5.11.0",
|
||||
"msgpack>=1.0.8",
|
||||
"slack-sdk>=3.26.0",
|
||||
"qq-botpy>=1.0.0",
|
||||
"python-socks[asyncio]>=2.4.0",
|
||||
"prompt-toolkit>=3.0.0",
|
||||
"vncdotool>=1.0.0",
|
||||
]
|
||||
|
||||
[project.optional-dependencies]
|
||||
matrix = [
|
||||
"matrix-nio[e2e]>=0.25.2",
|
||||
"mistune>=3.0.0,<4.0.0",
|
||||
"nh3>=0.2.17,<1.0.0",
|
||||
]
|
||||
dev = [
|
||||
"pytest>=9.0.0,<10.0.0",
|
||||
"pytest-asyncio>=1.3.0,<2.0.0",
|
||||
"pytest>=7.0.0",
|
||||
"pytest-asyncio>=0.21.0",
|
||||
"ruff>=0.1.0",
|
||||
]
|
||||
mem0 = [
|
||||
"mem0ai>=0.1.0",
|
||||
]
|
||||
matrix = [
|
||||
"matrix-nio>=0.20.0",
|
||||
"mistune>=3.0.0",
|
||||
"nh3>=0.2.0",
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
nanobot = "nanobot.cli.commands:app"
|
||||
@@ -69,11 +69,10 @@ packages = ["nanobot"]
|
||||
[tool.hatch.build.targets.wheel.sources]
|
||||
"nanobot" = "nanobot"
|
||||
|
||||
# Include non-Python files in skills and templates
|
||||
# Include non-Python files in skills
|
||||
[tool.hatch.build]
|
||||
include = [
|
||||
"nanobot/**/*.py",
|
||||
"nanobot/templates/**/*.md",
|
||||
"nanobot/skills/**/*.md",
|
||||
"nanobot/skills/**/*.sh",
|
||||
]
|
||||
|
||||
Executable
+34
@@ -0,0 +1,34 @@
|
||||
#!/bin/bash
|
||||
# test-pr.sh - Quick PR testing script for nanobot staging
|
||||
#
|
||||
# Usage: ./test-pr.sh <pr-number> [test-message]
|
||||
# Example: ./test-pr.sh 31 "test tool use feature"
|
||||
|
||||
set -e
|
||||
|
||||
PR_NUM="$1"
|
||||
TEST_MSG="${2:-Hello, testing PR #$PR_NUM}"
|
||||
REPO_DIR="/config/workspace/nanobot-oauth-port/nanobot-fork"
|
||||
STAGING_CONFIG="/config/workspace/.nanobot-staging/config.json"
|
||||
|
||||
if [ -z "$PR_NUM" ]; then
|
||||
echo "Usage: $0 <pr-number> [test-message]"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "==> Fetching PR #$PR_NUM..."
|
||||
cd "$REPO_DIR"
|
||||
git fetch wylab "+pull/$PR_NUM/head:pr-$PR_NUM"
|
||||
|
||||
echo "==> Checking out pr-$PR_NUM..."
|
||||
git checkout "pr-$PR_NUM"
|
||||
|
||||
echo "==> Installing in editable mode..."
|
||||
uv pip install -e . -q
|
||||
|
||||
echo "==> Testing with message: $TEST_MSG"
|
||||
NANOBOT_CONFIG="$STAGING_CONFIG" "$REPO_DIR/.venv/bin/nanobot" agent -m "$TEST_MSG"
|
||||
|
||||
echo ""
|
||||
echo "==> Test complete. Branch pr-$PR_NUM is still checked out."
|
||||
echo " Run 'git checkout main' to return to main branch."
|
||||
@@ -0,0 +1,102 @@
|
||||
# tests/test_agent_loop_metadata.py
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.providers.base import LLMProvider, LLMResponse
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_process_direct_passes_metadata():
|
||||
"""Test that process_direct passes metadata to InboundMessage."""
|
||||
bus = MessageBus()
|
||||
provider = MagicMock(spec=LLMProvider)
|
||||
provider.chat = AsyncMock(return_value=LLMResponse(
|
||||
content="test response",
|
||||
tool_calls=[]
|
||||
))
|
||||
provider.get_default_model = MagicMock(return_value="test-model")
|
||||
provider.thinking_budget = 0
|
||||
|
||||
workspace = Path("/tmp/test-workspace")
|
||||
workspace.mkdir(exist_ok=True)
|
||||
|
||||
loop = AgentLoop(bus=bus, provider=provider, workspace=workspace)
|
||||
|
||||
# Call with metadata
|
||||
test_metadata = {"suppress_output": True, "test_key": "test_value"}
|
||||
await loop.process_direct(
|
||||
content="test message",
|
||||
metadata=test_metadata
|
||||
)
|
||||
|
||||
# Verify provider.chat was called
|
||||
assert provider.chat.called
|
||||
call_args = provider.chat.call_args
|
||||
messages = call_args.kwargs["messages"]
|
||||
|
||||
# The user message should contain the content
|
||||
# (We can't easily check InboundMessage directly, but we verify
|
||||
# the flow worked by checking the session was created)
|
||||
session = loop.sessions.get_or_create("cli:direct")
|
||||
assert len(session.messages) > 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_suppress_mode_adds_hidden_prefix():
|
||||
"""Test that suppress_output metadata adds [HIDDEN:signature] prefix."""
|
||||
bus = MessageBus()
|
||||
provider = MagicMock(spec=LLMProvider)
|
||||
provider.chat = AsyncMock(return_value=LLMResponse(
|
||||
content="This is the agent response",
|
||||
tool_calls=[]
|
||||
))
|
||||
provider.get_default_model = MagicMock(return_value="test-model")
|
||||
provider.thinking_budget = 0
|
||||
|
||||
workspace = Path("/tmp/test-workspace")
|
||||
workspace.mkdir(exist_ok=True)
|
||||
|
||||
loop = AgentLoop(bus=bus, provider=provider, workspace=workspace)
|
||||
|
||||
# Call with suppress_output=True
|
||||
response = await loop.process_direct(
|
||||
content="test message",
|
||||
metadata={"suppress_output": True}
|
||||
)
|
||||
|
||||
# Response content should have [HIDDEN:signature] prefix with 8-char hex signature
|
||||
assert response.startswith("[HIDDEN:")
|
||||
assert "]" in response
|
||||
# Extract signature part between [HIDDEN: and ]
|
||||
prefix_end = response.index("]")
|
||||
signature = response[8:prefix_end] # Skip "[HIDDEN:" to get signature
|
||||
assert len(signature) == 8 # 8-character hex signature
|
||||
assert all(c in "0123456789abcdef" for c in signature) # Valid hex
|
||||
assert "This is the agent response" in response
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_normal_mode_no_hidden_prefix():
|
||||
"""Test that normal messages don't get [HIDDEN:signature] prefix."""
|
||||
bus = MessageBus()
|
||||
provider = MagicMock(spec=LLMProvider)
|
||||
provider.chat = AsyncMock(return_value=LLMResponse(
|
||||
content="Normal response",
|
||||
tool_calls=[]
|
||||
))
|
||||
provider.get_default_model = MagicMock(return_value="test-model")
|
||||
provider.thinking_budget = 0
|
||||
|
||||
workspace = Path("/tmp/test-workspace")
|
||||
workspace.mkdir(exist_ok=True)
|
||||
|
||||
loop = AgentLoop(bus=bus, provider=provider, workspace=workspace)
|
||||
|
||||
# Call without suppress_output
|
||||
response = await loop.process_direct(content="test message")
|
||||
|
||||
# Response should NOT have [HIDDEN:signature] prefix
|
||||
assert not response.startswith("[HIDDEN:")
|
||||
assert response == "Normal response"
|
||||
@@ -0,0 +1,262 @@
|
||||
"""Tests for agent loop handling of ToolResult and CLIResult objects."""
|
||||
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.agent.tools.anthropic.base import ToolResult, CLIResult
|
||||
from nanobot.bus.events import InboundMessage, OutboundMessage
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.providers.base import LLMResponse, ToolCallRequest
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_provider():
|
||||
"""Create mock LLM provider."""
|
||||
provider = MagicMock()
|
||||
provider.chat = AsyncMock()
|
||||
provider.thinking_budget = 0
|
||||
return provider
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_session_manager():
|
||||
"""Create mock session manager."""
|
||||
session_mgr = MagicMock()
|
||||
session_mgr.load = AsyncMock(return_value={
|
||||
"messages": [],
|
||||
"metadata": {},
|
||||
})
|
||||
session_mgr.save = MagicMock() # Synchronous in production, not async
|
||||
return session_mgr
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_bus():
|
||||
"""Create mock message bus."""
|
||||
bus = MagicMock(spec=MessageBus)
|
||||
bus.publish = AsyncMock()
|
||||
return bus
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def agent_loop(mock_provider, mock_session_manager, mock_bus, tmp_path):
|
||||
"""Create agent loop for testing."""
|
||||
return AgentLoop(
|
||||
provider=mock_provider,
|
||||
session_manager=mock_session_manager,
|
||||
bus=mock_bus,
|
||||
workspace=tmp_path,
|
||||
max_iterations=5,
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_tool_result_with_output(agent_loop, mock_provider):
|
||||
"""Test handling ToolResult with output field."""
|
||||
# Mock LLM responses
|
||||
mock_provider.chat.side_effect = [
|
||||
# First call: request tool
|
||||
LLMResponse(
|
||||
content="Using tool",
|
||||
tool_calls=[ToolCallRequest(id="call_1", name="test_tool", arguments={})],
|
||||
),
|
||||
# Second call: final response
|
||||
LLMResponse(content="Done"),
|
||||
]
|
||||
|
||||
# Mock tool that returns ToolResult
|
||||
tool_result = ToolResult(output="Tool executed successfully")
|
||||
agent_loop.tools.execute = AsyncMock(return_value=tool_result)
|
||||
|
||||
message = InboundMessage(
|
||||
channel="test",
|
||||
chat_id="123",
|
||||
sender_id="user1",
|
||||
content="Test message",
|
||||
)
|
||||
|
||||
response = await agent_loop._process_message(message)
|
||||
|
||||
# Verify tool result was added to messages
|
||||
calls = mock_provider.chat.call_args_list
|
||||
second_call_messages = calls[1][1]["messages"]
|
||||
|
||||
# Find the tool result message
|
||||
tool_msg = next(m for m in second_call_messages if m.get("role") == "tool")
|
||||
assert tool_msg["content"] == "Tool executed successfully"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_tool_result_with_error(agent_loop, mock_provider):
|
||||
"""Test handling ToolResult with error field."""
|
||||
mock_provider.chat.side_effect = [
|
||||
LLMResponse(
|
||||
content="Using tool",
|
||||
tool_calls=[ToolCallRequest(id="call_1", name="test_tool", arguments={})],
|
||||
),
|
||||
LLMResponse(content="Error handled"),
|
||||
]
|
||||
|
||||
tool_result = ToolResult(error="Command failed: exit code 1")
|
||||
agent_loop.tools.execute = AsyncMock(return_value=tool_result)
|
||||
|
||||
message = InboundMessage(
|
||||
channel="test",
|
||||
chat_id="123",
|
||||
sender_id="user1",
|
||||
content="Test message",
|
||||
)
|
||||
|
||||
response = await agent_loop._process_message(message)
|
||||
|
||||
calls = mock_provider.chat.call_args_list
|
||||
second_call_messages = calls[1][1]["messages"]
|
||||
tool_msg = next(m for m in second_call_messages if m.get("role") == "tool")
|
||||
assert "Error:" in tool_msg["content"]
|
||||
assert "Command failed: exit code 1" in tool_msg["content"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_tool_result_with_base64_image(agent_loop, mock_provider):
|
||||
"""Test handling ToolResult with base64_image field."""
|
||||
mock_provider.chat.side_effect = [
|
||||
LLMResponse(
|
||||
content="Taking screenshot",
|
||||
tool_calls=[ToolCallRequest(id="call_1", name="screenshot", arguments={})],
|
||||
),
|
||||
LLMResponse(content="Screenshot analyzed"),
|
||||
]
|
||||
|
||||
tool_result = ToolResult(
|
||||
output="Screenshot taken",
|
||||
base64_image="iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNk+M9QDwADhgGAWjR9awAAAABJRU5ErkJggg==",
|
||||
)
|
||||
agent_loop.tools.execute = AsyncMock(return_value=tool_result)
|
||||
|
||||
message = InboundMessage(
|
||||
channel="test",
|
||||
chat_id="123",
|
||||
sender_id="user1",
|
||||
content="Test message",
|
||||
)
|
||||
|
||||
response = await agent_loop._process_message(message)
|
||||
|
||||
calls = mock_provider.chat.call_args_list
|
||||
second_call_messages = calls[1][1]["messages"]
|
||||
tool_msg = next(m for m in second_call_messages if m.get("role") == "tool")
|
||||
|
||||
# Should contain both text and image
|
||||
assert isinstance(tool_msg["content"], list)
|
||||
assert len(tool_msg["content"]) == 2
|
||||
|
||||
# Text content
|
||||
text_part = next(p for p in tool_msg["content"] if p["type"] == "text")
|
||||
assert text_part["text"] == "Screenshot taken"
|
||||
|
||||
# Image content
|
||||
image_part = next(p for p in tool_msg["content"] if p["type"] == "image")
|
||||
assert image_part["source"]["type"] == "base64"
|
||||
assert image_part["source"]["media_type"] == "image/png"
|
||||
assert "iVBORw0KGgoAAAANS" in image_part["source"]["data"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_cli_result_handling(agent_loop, mock_provider):
|
||||
"""Test handling CLIResult from text editor tools."""
|
||||
mock_provider.chat.side_effect = [
|
||||
LLMResponse(
|
||||
content="Editing file",
|
||||
tool_calls=[ToolCallRequest(id="call_1", name="edit", arguments={})],
|
||||
),
|
||||
LLMResponse(content="File edited"),
|
||||
]
|
||||
|
||||
cli_result = CLIResult(
|
||||
exit_code=0,
|
||||
output="File updated successfully",
|
||||
error="",
|
||||
)
|
||||
agent_loop.tools.execute = AsyncMock(return_value=cli_result)
|
||||
|
||||
message = InboundMessage(
|
||||
channel="test",
|
||||
chat_id="123",
|
||||
sender_id="user1",
|
||||
content="Test message",
|
||||
)
|
||||
|
||||
response = await agent_loop._process_message(message)
|
||||
|
||||
calls = mock_provider.chat.call_args_list
|
||||
second_call_messages = calls[1][1]["messages"]
|
||||
tool_msg = next(m for m in second_call_messages if m.get("role") == "tool")
|
||||
assert tool_msg["content"] == "File updated successfully"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_legacy_string_result(agent_loop, mock_provider):
|
||||
"""Test backward compatibility with string results from function tools."""
|
||||
mock_provider.chat.side_effect = [
|
||||
LLMResponse(
|
||||
content="Using tool",
|
||||
tool_calls=[ToolCallRequest(id="call_1", name="legacy_tool", arguments={})],
|
||||
),
|
||||
LLMResponse(content="Done"),
|
||||
]
|
||||
|
||||
# Legacy tool returns plain string
|
||||
agent_loop.tools.execute = AsyncMock(return_value="Plain text result")
|
||||
|
||||
message = InboundMessage(
|
||||
channel="test",
|
||||
chat_id="123",
|
||||
sender_id="user1",
|
||||
content="Test message",
|
||||
)
|
||||
|
||||
response = await agent_loop._process_message(message)
|
||||
|
||||
calls = mock_provider.chat.call_args_list
|
||||
second_call_messages = calls[1][1]["messages"]
|
||||
tool_msg = next(m for m in second_call_messages if m.get("role") == "tool")
|
||||
assert tool_msg["content"] == "Plain text result"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_tool_result_output_and_error(agent_loop, mock_provider):
|
||||
"""Test handling ToolResult with both output and error."""
|
||||
mock_provider.chat.side_effect = [
|
||||
LLMResponse(
|
||||
content="Running command",
|
||||
tool_calls=[ToolCallRequest(id="call_1", name="bash", arguments={})],
|
||||
),
|
||||
LLMResponse(content="Handled"),
|
||||
]
|
||||
|
||||
tool_result = ToolResult(
|
||||
output="Partial output before error",
|
||||
error="Unexpected termination",
|
||||
)
|
||||
agent_loop.tools.execute = AsyncMock(return_value=tool_result)
|
||||
|
||||
message = InboundMessage(
|
||||
channel="test",
|
||||
chat_id="123",
|
||||
sender_id="user1",
|
||||
content="Test message",
|
||||
)
|
||||
|
||||
response = await agent_loop._process_message(message)
|
||||
|
||||
calls = mock_provider.chat.call_args_list
|
||||
second_call_messages = calls[1][1]["messages"]
|
||||
tool_msg = next(m for m in second_call_messages if m.get("role") == "tool")
|
||||
|
||||
# Should contain both output and error
|
||||
content = tool_msg["content"]
|
||||
assert "Partial output before error" in content
|
||||
assert "Error:" in content
|
||||
assert "Unexpected termination" in content
|
||||
@@ -0,0 +1,62 @@
|
||||
"""Tests for Anthropic native tool base classes."""
|
||||
|
||||
import pytest
|
||||
from nanobot.agent.tools.anthropic.base import (
|
||||
BaseAnthropicTool,
|
||||
ToolResult,
|
||||
CLIResult,
|
||||
ToolError,
|
||||
)
|
||||
|
||||
|
||||
class DummyTool(BaseAnthropicTool):
|
||||
"""Test tool implementation."""
|
||||
api_type = "test_20250227"
|
||||
name = "test_tool"
|
||||
beta_flag = "test-beta"
|
||||
|
||||
async def __call__(self, **kwargs):
|
||||
return ToolResult(output="test output")
|
||||
|
||||
def to_params(self):
|
||||
return {"type": self.api_type, "name": self.name}
|
||||
|
||||
|
||||
def test_tool_result_dataclass():
|
||||
"""Test ToolResult can be created with all fields."""
|
||||
result = ToolResult(output="hello", error=None, base64_image=None, system="system message")
|
||||
assert result.output == "hello"
|
||||
assert result.error is None
|
||||
assert result.base64_image is None
|
||||
assert result.system == "system message"
|
||||
|
||||
|
||||
def test_cli_result_dataclass():
|
||||
"""Test CLIResult can be created with all fields."""
|
||||
result = CLIResult(exit_code=0, output="command output", error="")
|
||||
assert result.output == "command output"
|
||||
assert result.exit_code == 0
|
||||
assert result.error == ""
|
||||
|
||||
|
||||
def test_tool_error_exception():
|
||||
"""Test ToolError can be raised and caught."""
|
||||
with pytest.raises(ToolError):
|
||||
raise ToolError("Test error message")
|
||||
|
||||
|
||||
def test_base_anthropic_tool_to_params():
|
||||
"""Test tool returns correct params format."""
|
||||
tool = DummyTool()
|
||||
params = tool.to_params()
|
||||
assert params["type"] == "test_20250227"
|
||||
assert params["name"] == "test_tool"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_base_anthropic_tool_call():
|
||||
"""Test tool can be called and returns ToolResult."""
|
||||
tool = DummyTool()
|
||||
result = await tool()
|
||||
assert isinstance(result, ToolResult)
|
||||
assert result.output == "test output"
|
||||
@@ -0,0 +1,78 @@
|
||||
"""Test Anthropic OAuth provider."""
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, patch, MagicMock
|
||||
from nanobot.providers.anthropic_oauth import AnthropicOAuthProvider
|
||||
from nanobot.providers.base import LLMResponse
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def provider():
|
||||
"""Create provider with test OAuth token."""
|
||||
return AnthropicOAuthProvider(
|
||||
oauth_token="sk-ant-oat01-test-token",
|
||||
default_model="claude-opus-4-7"
|
||||
)
|
||||
|
||||
|
||||
def test_provider_init(provider):
|
||||
"""Provider should initialize with OAuth token."""
|
||||
assert provider.oauth_token == "sk-ant-oat01-test-token"
|
||||
assert provider.default_model == "claude-opus-4-7"
|
||||
|
||||
|
||||
def test_provider_uses_bearer_auth(provider):
|
||||
"""Provider should use Bearer auth, not x-api-key."""
|
||||
headers = provider._get_headers()
|
||||
assert "Authorization" in headers
|
||||
assert headers["Authorization"].startswith("Bearer ")
|
||||
assert "x-api-key" not in headers
|
||||
|
||||
|
||||
# test_chat_prepends_system_prompt removed - feature no longer exists
|
||||
# System prompt handling is done by the agent loop, not the provider
|
||||
|
||||
|
||||
def test_parse_response_text(provider):
|
||||
"""Should parse text response correctly."""
|
||||
response = {
|
||||
"content": [{"type": "text", "text": "Hello world"}],
|
||||
"stop_reason": "end_turn",
|
||||
"usage": {"input_tokens": 10, "output_tokens": 5},
|
||||
}
|
||||
result = provider._parse_response(response)
|
||||
assert result.content == "Hello world"
|
||||
assert result.finish_reason == "end_turn"
|
||||
assert result.usage["prompt_tokens"] == 10
|
||||
|
||||
|
||||
def test_parse_response_tool_calls(provider):
|
||||
"""Should parse tool call response correctly."""
|
||||
response = {
|
||||
"content": [
|
||||
{"type": "tool_use", "id": "call_1", "name": "read_file", "input": {"path": "/tmp/test"}}
|
||||
],
|
||||
"stop_reason": "tool_use",
|
||||
"usage": {"input_tokens": 10, "output_tokens": 5},
|
||||
}
|
||||
result = provider._parse_response(response)
|
||||
assert len(result.tool_calls) == 1
|
||||
assert result.tool_calls[0].name == "read_file"
|
||||
assert result.tool_calls[0].arguments == {"path": "/tmp/test"}
|
||||
|
||||
|
||||
def test_convert_tools_to_anthropic(provider):
|
||||
"""Should convert OpenAI-format tools to Anthropic format."""
|
||||
openai_tools = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "read_file",
|
||||
"description": "Read a file",
|
||||
"parameters": {"type": "object", "properties": {"path": {"type": "string"}}}
|
||||
}
|
||||
}
|
||||
]
|
||||
anthropic_tools = provider._convert_tools_to_anthropic(openai_tools)
|
||||
assert len(anthropic_tools) == 1
|
||||
assert anthropic_tools[0]["name"] == "read_file"
|
||||
assert "input_schema" in anthropic_tools[0]
|
||||
@@ -0,0 +1,74 @@
|
||||
"""Tests for native tool support in AnthropicOAuthProvider."""
|
||||
|
||||
from nanobot.providers.anthropic_oauth import AnthropicOAuthProvider
|
||||
|
||||
|
||||
def test_convert_tools_passes_through_native_tools():
|
||||
"""Test that native tool format is passed through unchanged."""
|
||||
provider = AnthropicOAuthProvider(oauth_token="test", thinking_budget=0)
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "bash_20250124",
|
||||
"name": "bash"
|
||||
}
|
||||
]
|
||||
|
||||
result = provider._convert_tools_to_anthropic(tools)
|
||||
assert len(result) == 1
|
||||
assert result[0]["type"] == "bash_20250124"
|
||||
assert result[0]["name"] == "bash"
|
||||
|
||||
|
||||
def test_convert_tools_handles_mixed_tool_types():
|
||||
"""Test conversion of both function and native tools."""
|
||||
provider = AnthropicOAuthProvider(oauth_token="test", thinking_budget=0)
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "custom_tool",
|
||||
"description": "A custom tool",
|
||||
"parameters": {"type": "object", "properties": {"arg": {"type": "string"}}}
|
||||
}
|
||||
},
|
||||
{
|
||||
"type": "bash_20250124",
|
||||
"name": "bash"
|
||||
}
|
||||
]
|
||||
|
||||
result = provider._convert_tools_to_anthropic(tools)
|
||||
assert len(result) == 2
|
||||
|
||||
# Function tool gets converted
|
||||
assert result[0]["name"] == "custom_tool"
|
||||
assert result[0]["description"] == "A custom tool"
|
||||
assert "input_schema" in result[0]
|
||||
|
||||
# Native tool passed through
|
||||
assert result[1]["type"] == "bash_20250124"
|
||||
assert result[1]["name"] == "bash"
|
||||
|
||||
|
||||
def test_convert_tools_preserves_function_tool_conversion():
|
||||
"""Test that existing function tool conversion still works."""
|
||||
provider = AnthropicOAuthProvider(oauth_token="test", thinking_budget=0)
|
||||
|
||||
tools = [
|
||||
{
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": "test",
|
||||
"description": "desc",
|
||||
"parameters": {"type": "object"}
|
||||
}
|
||||
}
|
||||
]
|
||||
|
||||
result = provider._convert_tools_to_anthropic(tools)
|
||||
assert len(result) == 1
|
||||
assert result[0]["name"] == "test"
|
||||
assert result[0]["description"] == "desc"
|
||||
assert result[0]["input_schema"] == {"type": "object"}
|
||||
@@ -0,0 +1,68 @@
|
||||
"""Test that bash tool handles heredoc commands correctly.
|
||||
|
||||
Reproduces the bug where `; echo '<<exit>>'` appended on the same line
|
||||
as a heredoc terminator prevents bash from recognizing the terminator,
|
||||
causing the session to hang forever.
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import pytest
|
||||
from nanobot.agent.tools.anthropic.bash import BashTool20250124
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_heredoc_command():
|
||||
"""Heredoc commands must complete without hanging."""
|
||||
tool = BashTool20250124()
|
||||
|
||||
# Simple command works
|
||||
result = await tool(command="echo hello")
|
||||
assert result.output == "hello"
|
||||
|
||||
# Heredoc command — this is the exact pattern that caused the hang
|
||||
result = await asyncio.wait_for(
|
||||
tool(command="cat << 'EOF'\nline1\nline2\nEOF"),
|
||||
timeout=5.0,
|
||||
)
|
||||
assert "line1" in result.output
|
||||
assert "line2" in result.output
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_heredoc_append_to_file():
|
||||
"""Heredoc append (the exact pattern the LLM uses) must work."""
|
||||
tool = BashTool20250124()
|
||||
|
||||
result = await asyncio.wait_for(
|
||||
tool(command="cat >> /tmp/test_heredoc_bash.txt << 'EOF'\nhello world\nEOF"),
|
||||
timeout=5.0,
|
||||
)
|
||||
# Should complete without error
|
||||
assert result.error is None or result.error == ""
|
||||
|
||||
# Verify the file was written
|
||||
result2 = await tool(command="cat /tmp/test_heredoc_bash.txt")
|
||||
assert "hello world" in result2.output
|
||||
|
||||
# Cleanup
|
||||
await tool(command="rm -f /tmp/test_heredoc_bash.txt")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_regular_commands_still_work():
|
||||
"""Ensure regular commands still work after the fix."""
|
||||
tool = BashTool20250124()
|
||||
|
||||
# Semicolons in commands
|
||||
result = await tool(command="echo a; echo b")
|
||||
assert "a" in result.output
|
||||
assert "b" in result.output
|
||||
|
||||
# Multiline script
|
||||
result = await tool(command="for i in 1 2 3; do echo $i; done")
|
||||
assert "1" in result.output
|
||||
assert "3" in result.output
|
||||
|
||||
# Command with exit code
|
||||
result = await tool(command="true")
|
||||
assert result.output == "(no output)" or result.output is not None
|
||||
@@ -0,0 +1,56 @@
|
||||
"""Tests for BashTool20250124."""
|
||||
|
||||
import pytest
|
||||
from nanobot.agent.tools.anthropic.bash import BashTool20250124
|
||||
from nanobot.agent.tools.anthropic.base import ToolResult
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_bash_tool_simple_command():
|
||||
"""Test bash tool executes simple command."""
|
||||
tool = BashTool20250124()
|
||||
result = await tool(command="echo hello")
|
||||
|
||||
assert isinstance(result, ToolResult)
|
||||
assert "hello" in result.output
|
||||
assert result.error is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_bash_tool_persistent_session():
|
||||
"""Test bash tool maintains session across calls."""
|
||||
tool = BashTool20250124()
|
||||
|
||||
# Set variable
|
||||
result1 = await tool(command="export TEST_VAR=42")
|
||||
assert result1.error is None
|
||||
|
||||
# Read variable (should persist)
|
||||
result2 = await tool(command="echo $TEST_VAR")
|
||||
assert "42" in result2.output
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_bash_tool_restart():
|
||||
"""Test bash tool can restart session."""
|
||||
tool = BashTool20250124()
|
||||
|
||||
# Set variable
|
||||
await tool(command="export TEST_VAR=42")
|
||||
|
||||
# Restart
|
||||
result = await tool(restart=True)
|
||||
assert "restarted" in (result.system or result.output or "").lower()
|
||||
|
||||
# Variable should be gone
|
||||
result2 = await tool(command="echo $TEST_VAR")
|
||||
assert "42" not in result2.output
|
||||
|
||||
|
||||
def test_bash_tool_to_params():
|
||||
"""Test bash tool returns correct params."""
|
||||
tool = BashTool20250124()
|
||||
params = tool.to_params()
|
||||
|
||||
assert params["type"] == "bash_20250124"
|
||||
assert params["name"] == "bash"
|
||||
@@ -0,0 +1,104 @@
|
||||
"""Tests for beta flag collection from native tools."""
|
||||
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from nanobot.providers.anthropic_oauth import AnthropicOAuthProvider
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_beta_flags_collected_from_tools():
|
||||
"""Test that beta flags are extracted from tool objects."""
|
||||
provider = AnthropicOAuthProvider(oauth_token="test", thinking_budget=0)
|
||||
|
||||
# Mock tool objects with beta_flag attribute and to_params method
|
||||
class MockTool:
|
||||
def __init__(self, beta_flag):
|
||||
self.beta_flag = beta_flag
|
||||
|
||||
def to_params(self):
|
||||
return {"type": "bash_20250124", "name": "bash"}
|
||||
|
||||
tools_with_flags = [
|
||||
MockTool("computer-use-2025-11-24"),
|
||||
MockTool("computer-use-2025-11-24"), # Duplicate should be deduplicated
|
||||
]
|
||||
|
||||
# We need to test this via the actual API call flow
|
||||
# Mock httpx client
|
||||
mock_response = MagicMock()
|
||||
mock_response.status_code = 200
|
||||
mock_response.json.return_value = {
|
||||
"id": "msg_test",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"content": [{"type": "text", "text": "test"}],
|
||||
"model": "claude-opus-4",
|
||||
"stop_reason": "end_turn",
|
||||
"usage": {"input_tokens": 10, "output_tokens": 10}
|
||||
}
|
||||
|
||||
with patch.object(provider, '_client') as mock_client:
|
||||
mock_client.post = AsyncMock(return_value=mock_response)
|
||||
|
||||
# Call with messages and tools
|
||||
await provider.chat(
|
||||
messages=[{"role": "user", "content": "test"}],
|
||||
model="claude-opus-4",
|
||||
max_tokens=100,
|
||||
tools=tools_with_flags
|
||||
)
|
||||
|
||||
# Check that beta flag was added to headers (merged with hardcoded flags)
|
||||
call_args = mock_client.post.call_args
|
||||
headers = call_args[1]["headers"]
|
||||
assert "anthropic-beta" in headers
|
||||
# Should include hardcoded flags + tool flag, sorted alphabetically
|
||||
assert headers["anthropic-beta"] == "claude-code-20250219,computer-use-2025-11-24,context-management-2025-06-27,oauth-2025-04-20"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_multiple_beta_flags_joined():
|
||||
"""Test that multiple unique beta flags are joined with commas."""
|
||||
provider = AnthropicOAuthProvider(oauth_token="test", thinking_budget=0)
|
||||
|
||||
class MockTool:
|
||||
def __init__(self, beta_flag):
|
||||
self.beta_flag = beta_flag
|
||||
|
||||
def to_params(self):
|
||||
return {"type": "bash_20250124", "name": "bash"}
|
||||
|
||||
tools_with_flags = [
|
||||
MockTool("flag-a"),
|
||||
MockTool("flag-b"),
|
||||
]
|
||||
|
||||
mock_response = MagicMock()
|
||||
mock_response.status_code = 200
|
||||
mock_response.json.return_value = {
|
||||
"id": "msg_test",
|
||||
"type": "message",
|
||||
"role": "assistant",
|
||||
"content": [{"type": "text", "text": "test"}],
|
||||
"model": "claude-opus-4",
|
||||
"stop_reason": "end_turn",
|
||||
"usage": {"input_tokens": 10, "output_tokens": 10}
|
||||
}
|
||||
|
||||
with patch.object(provider, '_client') as mock_client:
|
||||
mock_client.post = AsyncMock(return_value=mock_response)
|
||||
|
||||
await provider.chat(
|
||||
messages=[{"role": "user", "content": "test"}],
|
||||
model="claude-opus-4",
|
||||
max_tokens=100,
|
||||
tools=tools_with_flags
|
||||
)
|
||||
|
||||
call_args = mock_client.post.call_args
|
||||
headers = call_args[1]["headers"]
|
||||
assert "anthropic-beta" in headers
|
||||
# Should include hardcoded flags + tool flags, sorted alphabetically and joined with comma
|
||||
assert headers["anthropic-beta"] == "claude-code-20250219,context-management-2025-06-27,flag-a,flag-b,oauth-2025-04-20"
|
||||
@@ -0,0 +1,59 @@
|
||||
"""Tests for bus-level correlation (request-response via Futures)."""
|
||||
|
||||
import asyncio
|
||||
import pytest
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.bus.events import OutboundMessage
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def bus():
|
||||
return MessageBus()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_register_correlation_returns_future(bus):
|
||||
future = bus.register_correlation("test-id-1")
|
||||
assert isinstance(future, asyncio.Future)
|
||||
assert not future.done()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_correlation_sets_future_result(bus):
|
||||
future = bus.register_correlation("test-id-1")
|
||||
msg = OutboundMessage(channel="hook", chat_id="test", content="hello", metadata={"correlation_id": "test-id-1"})
|
||||
bus.resolve_correlation(msg)
|
||||
assert future.done()
|
||||
assert future.result() == "hello"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_correlation_no_match_is_noop(bus):
|
||||
future = bus.register_correlation("test-id-1")
|
||||
msg = OutboundMessage(channel="hook", chat_id="test", content="hello", metadata={"correlation_id": "other-id"})
|
||||
bus.resolve_correlation(msg)
|
||||
assert not future.done()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_correlation_no_metadata_is_noop(bus):
|
||||
future = bus.register_correlation("test-id-1")
|
||||
msg = OutboundMessage(channel="hook", chat_id="test", content="hello")
|
||||
bus.resolve_correlation(msg)
|
||||
assert not future.done()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_resolve_correlation_cleans_up_store(bus):
|
||||
future = bus.register_correlation("test-id-1")
|
||||
msg = OutboundMessage(channel="hook", chat_id="test", content="hello", metadata={"correlation_id": "test-id-1"})
|
||||
bus.resolve_correlation(msg)
|
||||
assert "test-id-1" not in bus._correlation_store
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_cancel_correlation(bus):
|
||||
future = bus.register_correlation("test-id-1")
|
||||
bus.cancel_correlation("test-id-1")
|
||||
assert "test-id-1" not in bus._correlation_store
|
||||
assert future.cancelled()
|
||||
@@ -0,0 +1,55 @@
|
||||
"""Test OAuth CLI commands."""
|
||||
import pytest
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from typer.testing import CliRunner
|
||||
from nanobot.cli.oauth import oauth_app
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def runner():
|
||||
return CliRunner()
|
||||
|
||||
|
||||
def test_oauth_login_help(runner):
|
||||
"""Login command should have help text."""
|
||||
result = runner.invoke(oauth_app, ["login", "--help"])
|
||||
assert result.exit_code == 0
|
||||
assert "token" in result.output.lower()
|
||||
|
||||
|
||||
def test_oauth_status_no_credentials(runner, tmp_path, monkeypatch):
|
||||
"""Status should show no credentials when none exist."""
|
||||
monkeypatch.setenv("HOME", str(tmp_path))
|
||||
result = runner.invoke(oauth_app, ["status"])
|
||||
assert result.exit_code == 0
|
||||
assert "No OAuth credentials" in result.output
|
||||
|
||||
|
||||
def test_oauth_login_and_status(runner, tmp_path, monkeypatch):
|
||||
"""Login should save credentials, status should show them."""
|
||||
monkeypatch.setenv("HOME", str(tmp_path))
|
||||
result = runner.invoke(oauth_app, ["login", "--token", "sk-ant-oat01-test-xxx"])
|
||||
assert result.exit_code == 0
|
||||
assert "Successfully saved" in result.output
|
||||
|
||||
result = runner.invoke(oauth_app, ["status"])
|
||||
assert result.exit_code == 0
|
||||
assert "sk-ant-oat01-test-x" in result.output
|
||||
|
||||
|
||||
def test_oauth_logout(runner, tmp_path, monkeypatch):
|
||||
"""Logout should remove credentials."""
|
||||
monkeypatch.setenv("HOME", str(tmp_path))
|
||||
runner.invoke(oauth_app, ["login", "--token", "sk-ant-oat01-test-xxx"])
|
||||
result = runner.invoke(oauth_app, ["logout"])
|
||||
assert result.exit_code == 0
|
||||
assert "Removed" in result.output
|
||||
|
||||
|
||||
def test_oauth_login_invalid_token(runner, tmp_path, monkeypatch):
|
||||
"""Login should reject non-OAuth tokens."""
|
||||
monkeypatch.setenv("HOME", str(tmp_path))
|
||||
result = runner.invoke(oauth_app, ["login", "--token", "sk-ant-api03-regular"])
|
||||
assert result.exit_code == 0
|
||||
assert "Invalid token" in result.output
|
||||
+11
-11
@@ -29,6 +29,7 @@ def mock_paths():
|
||||
|
||||
config_file = base_dir / "config.json"
|
||||
workspace_dir = base_dir / "workspace"
|
||||
workspace_dir.mkdir() # Create workspace directory
|
||||
|
||||
mock_cp.return_value = config_file
|
||||
mock_ws.return_value = workspace_dir
|
||||
@@ -56,21 +57,20 @@ def test_onboard_fresh_install(mock_paths):
|
||||
|
||||
|
||||
def test_onboard_existing_config_refresh(mock_paths):
|
||||
"""Config exists, user declines overwrite — should refresh (load-merge-save)."""
|
||||
"""Config exists, user declines overwrite — should exit without changes."""
|
||||
config_file, workspace_dir = mock_paths
|
||||
config_file.write_text('{"existing": true}')
|
||||
|
||||
result = runner.invoke(app, ["onboard"], input="n\n")
|
||||
|
||||
# User declined, so command exits (typer.Exit() returns 0)
|
||||
assert result.exit_code == 0
|
||||
assert "Config already exists" in result.stdout
|
||||
assert "existing values preserved" in result.stdout
|
||||
assert workspace_dir.exists()
|
||||
assert (workspace_dir / "AGENTS.md").exists()
|
||||
assert "Overwrite?" in result.stdout
|
||||
|
||||
|
||||
def test_onboard_existing_config_overwrite(mock_paths):
|
||||
"""Config exists, user confirms overwrite — should reset to defaults."""
|
||||
"""Config exists, user confirms overwrite — should create new config."""
|
||||
config_file, workspace_dir = mock_paths
|
||||
config_file.write_text('{"existing": true}')
|
||||
|
||||
@@ -78,20 +78,20 @@ def test_onboard_existing_config_overwrite(mock_paths):
|
||||
|
||||
assert result.exit_code == 0
|
||||
assert "Config already exists" in result.stdout
|
||||
assert "Config reset to defaults" in result.stdout
|
||||
assert "Created config" in result.stdout
|
||||
assert workspace_dir.exists()
|
||||
|
||||
|
||||
def test_onboard_existing_workspace_safe_create(mock_paths):
|
||||
"""Workspace exists — should not recreate, but still add missing templates."""
|
||||
"""Workspace exists (from fixture) — should add missing templates."""
|
||||
config_file, workspace_dir = mock_paths
|
||||
workspace_dir.mkdir(parents=True)
|
||||
config_file.write_text("{}")
|
||||
# workspace_dir already exists from fixture
|
||||
# No existing config, so onboard should proceed
|
||||
|
||||
result = runner.invoke(app, ["onboard"], input="n\n")
|
||||
result = runner.invoke(app, ["onboard"])
|
||||
|
||||
assert result.exit_code == 0
|
||||
assert "Created workspace" not in result.stdout
|
||||
assert "Created workspace" in result.stdout
|
||||
assert "Created AGENTS.md" in result.stdout
|
||||
assert (workspace_dir / "AGENTS.md").exists()
|
||||
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
"""Tests for ComputerTool20251124."""
|
||||
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, patch, MagicMock
|
||||
from nanobot.agent.tools.anthropic.computer import ComputerTool20251124
|
||||
from nanobot.agent.tools.anthropic.base import ToolResult
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_computer_tool_screenshot():
|
||||
"""Test computer tool can take screenshot."""
|
||||
tool = ComputerTool20251124(vnc_host="localhost", vnc_port=5900)
|
||||
|
||||
# Mock VNC client
|
||||
with patch('nanobot.agent.tools.anthropic.computer.vnc_api.connect') as mock_connect:
|
||||
mock_client = MagicMock()
|
||||
# Mock captureScreen to write fake PNG data to file path
|
||||
def fake_capture(path):
|
||||
from pathlib import Path
|
||||
Path(path).write_bytes(b"fake_png_data")
|
||||
mock_client.captureScreen = MagicMock(side_effect=fake_capture)
|
||||
mock_client.mouseMove = MagicMock()
|
||||
mock_client.keyPress = MagicMock()
|
||||
mock_client.refreshScreen = MagicMock()
|
||||
mock_connect.return_value = mock_client
|
||||
|
||||
result = await tool(action="screenshot")
|
||||
|
||||
assert isinstance(result, ToolResult)
|
||||
assert result.base64_image is not None
|
||||
assert len(result.base64_image) > 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_computer_tool_mouse_move():
|
||||
"""Test computer tool can move mouse."""
|
||||
tool = ComputerTool20251124(vnc_host="localhost", vnc_port=5900)
|
||||
|
||||
with patch('nanobot.agent.tools.anthropic.computer.vnc_api.connect') as mock_connect:
|
||||
mock_client = MagicMock()
|
||||
mock_client.mouseMove = MagicMock()
|
||||
mock_connect.return_value = mock_client
|
||||
|
||||
result = await tool(action="mouse_move", coordinate=[100, 200])
|
||||
|
||||
assert isinstance(result, ToolResult)
|
||||
assert result.error is None
|
||||
mock_client.mouseMove.assert_called_once_with(100, 200)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_computer_tool_key():
|
||||
"""Test computer tool can press keys."""
|
||||
tool = ComputerTool20251124(vnc_host="localhost", vnc_port=5900)
|
||||
|
||||
with patch('nanobot.agent.tools.anthropic.computer.vnc_api.connect') as mock_connect:
|
||||
mock_client = MagicMock()
|
||||
mock_client.keyPress = MagicMock()
|
||||
mock_connect.return_value = mock_client
|
||||
|
||||
result = await tool(action="key", text="Return")
|
||||
|
||||
assert isinstance(result, ToolResult)
|
||||
assert result.error is None
|
||||
# Implementation converts keys to lowercase
|
||||
mock_client.keyPress.assert_called_once_with("return")
|
||||
|
||||
|
||||
def test_computer_tool_to_params():
|
||||
"""Test computer tool returns correct params."""
|
||||
tool = ComputerTool20251124(vnc_host="localhost", vnc_port=5900)
|
||||
params = tool.to_params()
|
||||
|
||||
assert params["type"] == "computer_20251124"
|
||||
assert params["name"] == "computer"
|
||||
@@ -0,0 +1,142 @@
|
||||
"""Tests for config loader (get_config_path and _migrate_config)"""
|
||||
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
from nanobot.config.loader import get_config_path, _migrate_config
|
||||
|
||||
|
||||
def test_get_config_path_default():
|
||||
"""get_config_path returns ~/.nanobot/config.json by default"""
|
||||
# Ensure NANOBOT_CONFIG is not set
|
||||
env_backup = os.environ.pop("NANOBOT_CONFIG", None)
|
||||
try:
|
||||
path = get_config_path()
|
||||
assert path == Path.home() / ".nanobot" / "config.json"
|
||||
finally:
|
||||
if env_backup:
|
||||
os.environ["NANOBOT_CONFIG"] = env_backup
|
||||
|
||||
|
||||
def test_get_config_path_with_env_var():
|
||||
"""get_config_path uses NANOBOT_CONFIG env var when set"""
|
||||
custom_path = "/tmp/test-nanobot-config.json"
|
||||
env_backup = os.environ.get("NANOBOT_CONFIG")
|
||||
try:
|
||||
os.environ["NANOBOT_CONFIG"] = custom_path
|
||||
path = get_config_path()
|
||||
assert path == Path(custom_path)
|
||||
finally:
|
||||
if env_backup:
|
||||
os.environ["NANOBOT_CONFIG"] = env_backup
|
||||
else:
|
||||
os.environ.pop("NANOBOT_CONFIG", None)
|
||||
|
||||
|
||||
def test_migrate_config_with_oauth_credentials():
|
||||
"""_migrate_config extracts api_key from oauthCredentials"""
|
||||
data = {
|
||||
"providers": {
|
||||
"anthropic": {
|
||||
"oauthCredentials": {
|
||||
"access_token": "sk-ant-test-token",
|
||||
"refresh_token": "",
|
||||
"expires_at": 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
result = _migrate_config(data)
|
||||
|
||||
# api_key should be extracted
|
||||
assert result["providers"]["anthropic"]["api_key"] == "sk-ant-test-token"
|
||||
# oauthCredentials should be removed after migration
|
||||
assert "oauthCredentials" not in result["providers"]["anthropic"]
|
||||
|
||||
|
||||
def test_migrate_config_without_oauth_credentials():
|
||||
"""_migrate_config leaves config unchanged when no oauthCredentials"""
|
||||
data = {
|
||||
"providers": {
|
||||
"anthropic": {
|
||||
"api_key": "sk-ant-existing-key"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
result = _migrate_config(data)
|
||||
|
||||
# Should remain unchanged
|
||||
assert result["providers"]["anthropic"]["api_key"] == "sk-ant-existing-key"
|
||||
assert "oauthCredentials" not in result["providers"]["anthropic"]
|
||||
|
||||
|
||||
def test_migrate_config_already_migrated():
|
||||
"""_migrate_config doesn't overwrite existing api_key"""
|
||||
data = {
|
||||
"providers": {
|
||||
"anthropic": {
|
||||
"api_key": "sk-ant-existing-key",
|
||||
"oauthCredentials": {
|
||||
"access_token": "sk-ant-oauth-token",
|
||||
"refresh_token": "",
|
||||
"expires_at": 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
result = _migrate_config(data)
|
||||
|
||||
# Existing api_key should be preserved
|
||||
assert result["providers"]["anthropic"]["api_key"] == "sk-ant-existing-key"
|
||||
# oauthCredentials should NOT be removed (api_key already existed)
|
||||
assert "oauthCredentials" in result["providers"]["anthropic"]
|
||||
|
||||
|
||||
def test_migrate_config_empty_access_token():
|
||||
"""_migrate_config skips empty access_token"""
|
||||
data = {
|
||||
"providers": {
|
||||
"anthropic": {
|
||||
"oauthCredentials": {
|
||||
"access_token": "",
|
||||
"refresh_token": "",
|
||||
"expires_at": 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
result = _migrate_config(data)
|
||||
|
||||
# api_key should not be set
|
||||
assert "api_key" not in result["providers"]["anthropic"]
|
||||
# oauthCredentials should remain (no migration happened)
|
||||
assert "oauthCredentials" in result["providers"]["anthropic"]
|
||||
|
||||
|
||||
def test_migrate_config_preserves_other_fields():
|
||||
"""_migrate_config preserves other provider config fields"""
|
||||
data = {
|
||||
"providers": {
|
||||
"anthropic": {
|
||||
"oauthCredentials": {
|
||||
"access_token": "sk-ant-test-token",
|
||||
"refresh_token": "refresh-token",
|
||||
},
|
||||
"customField": "customValue",
|
||||
"anotherField": 123,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
result = _migrate_config(data)
|
||||
|
||||
# api_key added, oauthCredentials removed
|
||||
assert result["providers"]["anthropic"]["api_key"] == "sk-ant-test-token"
|
||||
assert "oauthCredentials" not in result["providers"]["anthropic"]
|
||||
# Other fields preserved
|
||||
assert result["providers"]["anthropic"]["customField"] == "customValue"
|
||||
assert result["providers"]["anthropic"]["anotherField"] == 123
|
||||
@@ -0,0 +1,65 @@
|
||||
"""Test OAuth store integration with config loading."""
|
||||
import json
|
||||
import pytest
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from nanobot.config.loader import load_config
|
||||
from nanobot.config.oauth_store import OAuthStore
|
||||
from nanobot.config.schema import OAuthCredentials
|
||||
|
||||
|
||||
def test_oauth_token_injected_into_config(tmp_path, monkeypatch):
|
||||
"""OAuth token from store should be injected into provider api_key."""
|
||||
# Create a minimal config file (no api key set)
|
||||
config_path = tmp_path / "config.json"
|
||||
config_path.write_text(json.dumps({
|
||||
"agents": {"defaults": {"model": "anthropic/claude-opus-4-7"}},
|
||||
"providers": {"anthropic": {"apiKey": ""}}
|
||||
}))
|
||||
|
||||
# Save OAuth credentials
|
||||
store = OAuthStore(tmp_path)
|
||||
creds = OAuthCredentials(access_token="sk-ant-oat01-test-inject")
|
||||
store.save("anthropic", creds)
|
||||
|
||||
# Monkeypatch get_config_path to use our tmp dir
|
||||
monkeypatch.setattr("nanobot.config.loader.get_config_path", lambda: config_path)
|
||||
# Monkeypatch the OAuth store path
|
||||
monkeypatch.setattr("nanobot.config.loader._get_oauth_store_dir", lambda: tmp_path)
|
||||
|
||||
config = load_config(config_path)
|
||||
|
||||
assert config.providers.anthropic.api_key == "sk-ant-oat01-test-inject"
|
||||
|
||||
|
||||
def test_config_without_oauth_unchanged(tmp_path, monkeypatch):
|
||||
"""Config without OAuth store should load normally."""
|
||||
config_path = tmp_path / "config.json"
|
||||
config_path.write_text(json.dumps({
|
||||
"providers": {"anthropic": {"apiKey": "sk-ant-api03-regular"}}
|
||||
}))
|
||||
|
||||
monkeypatch.setattr("nanobot.config.loader._get_oauth_store_dir", lambda: tmp_path / "nonexistent")
|
||||
|
||||
config = load_config(config_path)
|
||||
|
||||
assert config.providers.anthropic.api_key == "sk-ant-api03-regular"
|
||||
|
||||
|
||||
def test_oauth_does_not_overwrite_existing_key(tmp_path, monkeypatch):
|
||||
"""If user already has an API key, OAuth should still override (OAuth takes priority)."""
|
||||
config_path = tmp_path / "config.json"
|
||||
config_path.write_text(json.dumps({
|
||||
"providers": {"anthropic": {"apiKey": "sk-ant-api03-existing"}}
|
||||
}))
|
||||
|
||||
store = OAuthStore(tmp_path)
|
||||
creds = OAuthCredentials(access_token="sk-ant-oat01-oauth-wins")
|
||||
store.save("anthropic", creds)
|
||||
|
||||
monkeypatch.setattr("nanobot.config.loader._get_oauth_store_dir", lambda: tmp_path)
|
||||
|
||||
config = load_config(config_path)
|
||||
|
||||
# OAuth token takes priority over existing API key
|
||||
assert config.providers.anthropic.api_key == "sk-ant-oat01-oauth-wins"
|
||||
@@ -1,828 +0,0 @@
|
||||
"""Test session management with cache-friendly message handling."""
|
||||
|
||||
import asyncio
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
from nanobot.session.manager import Session, SessionManager
|
||||
|
||||
# Test constants
|
||||
MEMORY_WINDOW = 50
|
||||
KEEP_COUNT = MEMORY_WINDOW // 2 # 25
|
||||
|
||||
|
||||
def create_session_with_messages(key: str, count: int, role: str = "user") -> Session:
|
||||
"""Create a session and add the specified number of messages.
|
||||
|
||||
Args:
|
||||
key: Session identifier
|
||||
count: Number of messages to add
|
||||
role: Message role (default: "user")
|
||||
|
||||
Returns:
|
||||
Session with the specified messages
|
||||
"""
|
||||
session = Session(key=key)
|
||||
for i in range(count):
|
||||
session.add_message(role, f"msg{i}")
|
||||
return session
|
||||
|
||||
|
||||
def assert_messages_content(messages: list, start_index: int, end_index: int) -> None:
|
||||
"""Assert that messages contain expected content from start to end index.
|
||||
|
||||
Args:
|
||||
messages: List of message dictionaries
|
||||
start_index: Expected first message index
|
||||
end_index: Expected last message index
|
||||
"""
|
||||
assert len(messages) > 0
|
||||
assert messages[0]["content"] == f"msg{start_index}"
|
||||
assert messages[-1]["content"] == f"msg{end_index}"
|
||||
|
||||
|
||||
def get_old_messages(session: Session, last_consolidated: int, keep_count: int) -> list:
|
||||
"""Extract messages that would be consolidated using the standard slice logic.
|
||||
|
||||
Args:
|
||||
session: The session containing messages
|
||||
last_consolidated: Index of last consolidated message
|
||||
keep_count: Number of recent messages to keep
|
||||
|
||||
Returns:
|
||||
List of messages that would be consolidated
|
||||
"""
|
||||
return session.messages[last_consolidated:-keep_count]
|
||||
|
||||
|
||||
class TestSessionLastConsolidated:
|
||||
"""Test last_consolidated tracking to avoid duplicate processing."""
|
||||
|
||||
def test_initial_last_consolidated_zero(self) -> None:
|
||||
"""Test that new session starts with last_consolidated=0."""
|
||||
session = Session(key="test:initial")
|
||||
assert session.last_consolidated == 0
|
||||
|
||||
def test_last_consolidated_persistence(self, tmp_path) -> None:
|
||||
"""Test that last_consolidated persists across save/load."""
|
||||
manager = SessionManager(Path(tmp_path))
|
||||
session1 = create_session_with_messages("test:persist", 20)
|
||||
session1.last_consolidated = 15
|
||||
manager.save(session1)
|
||||
|
||||
session2 = manager.get_or_create("test:persist")
|
||||
assert session2.last_consolidated == 15
|
||||
assert len(session2.messages) == 20
|
||||
|
||||
def test_clear_resets_last_consolidated(self) -> None:
|
||||
"""Test that clear() resets last_consolidated to 0."""
|
||||
session = create_session_with_messages("test:clear", 10)
|
||||
session.last_consolidated = 5
|
||||
|
||||
session.clear()
|
||||
assert len(session.messages) == 0
|
||||
assert session.last_consolidated == 0
|
||||
|
||||
|
||||
class TestSessionImmutableHistory:
|
||||
"""Test Session message immutability for cache efficiency."""
|
||||
|
||||
def test_initial_state(self) -> None:
|
||||
"""Test that new session has empty messages list."""
|
||||
session = Session(key="test:initial")
|
||||
assert len(session.messages) == 0
|
||||
|
||||
def test_add_messages_appends_only(self) -> None:
|
||||
"""Test that adding messages only appends, never modifies."""
|
||||
session = Session(key="test:preserve")
|
||||
session.add_message("user", "msg1")
|
||||
session.add_message("assistant", "resp1")
|
||||
session.add_message("user", "msg2")
|
||||
assert len(session.messages) == 3
|
||||
assert session.messages[0]["content"] == "msg1"
|
||||
|
||||
def test_get_history_returns_most_recent(self) -> None:
|
||||
"""Test get_history returns the most recent messages."""
|
||||
session = Session(key="test:history")
|
||||
for i in range(10):
|
||||
session.add_message("user", f"msg{i}")
|
||||
session.add_message("assistant", f"resp{i}")
|
||||
|
||||
history = session.get_history(max_messages=6)
|
||||
assert len(history) == 6
|
||||
assert history[0]["content"] == "msg7"
|
||||
assert history[-1]["content"] == "resp9"
|
||||
|
||||
def test_get_history_with_all_messages(self) -> None:
|
||||
"""Test get_history with max_messages larger than actual."""
|
||||
session = create_session_with_messages("test:all", 5)
|
||||
history = session.get_history(max_messages=100)
|
||||
assert len(history) == 5
|
||||
assert history[0]["content"] == "msg0"
|
||||
|
||||
def test_get_history_stable_for_same_session(self) -> None:
|
||||
"""Test that get_history returns same content for same max_messages."""
|
||||
session = create_session_with_messages("test:stable", 20)
|
||||
history1 = session.get_history(max_messages=10)
|
||||
history2 = session.get_history(max_messages=10)
|
||||
assert history1 == history2
|
||||
|
||||
def test_messages_list_never_modified(self) -> None:
|
||||
"""Test that messages list is never modified after creation."""
|
||||
session = create_session_with_messages("test:immutable", 5)
|
||||
original_len = len(session.messages)
|
||||
|
||||
session.get_history(max_messages=2)
|
||||
assert len(session.messages) == original_len
|
||||
|
||||
for _ in range(10):
|
||||
session.get_history(max_messages=3)
|
||||
assert len(session.messages) == original_len
|
||||
|
||||
|
||||
class TestSessionPersistence:
|
||||
"""Test Session persistence and reload."""
|
||||
|
||||
@pytest.fixture
|
||||
def temp_manager(self, tmp_path):
|
||||
return SessionManager(Path(tmp_path))
|
||||
|
||||
def test_persistence_roundtrip(self, temp_manager):
|
||||
"""Test that messages persist across save/load."""
|
||||
session1 = create_session_with_messages("test:persistence", 20)
|
||||
temp_manager.save(session1)
|
||||
|
||||
session2 = temp_manager.get_or_create("test:persistence")
|
||||
assert len(session2.messages) == 20
|
||||
assert session2.messages[0]["content"] == "msg0"
|
||||
assert session2.messages[-1]["content"] == "msg19"
|
||||
|
||||
def test_get_history_after_reload(self, temp_manager):
|
||||
"""Test that get_history works correctly after reload."""
|
||||
session1 = create_session_with_messages("test:reload", 30)
|
||||
temp_manager.save(session1)
|
||||
|
||||
session2 = temp_manager.get_or_create("test:reload")
|
||||
history = session2.get_history(max_messages=10)
|
||||
assert len(history) == 10
|
||||
assert history[0]["content"] == "msg20"
|
||||
assert history[-1]["content"] == "msg29"
|
||||
|
||||
def test_clear_resets_session(self, temp_manager):
|
||||
"""Test that clear() properly resets session."""
|
||||
session = create_session_with_messages("test:clear", 10)
|
||||
assert len(session.messages) == 10
|
||||
|
||||
session.clear()
|
||||
assert len(session.messages) == 0
|
||||
|
||||
|
||||
class TestConsolidationTriggerConditions:
|
||||
"""Test consolidation trigger conditions and logic."""
|
||||
|
||||
def test_consolidation_needed_when_messages_exceed_window(self):
|
||||
"""Test consolidation logic: should trigger when messages > memory_window."""
|
||||
session = create_session_with_messages("test:trigger", 60)
|
||||
|
||||
total_messages = len(session.messages)
|
||||
messages_to_process = total_messages - session.last_consolidated
|
||||
|
||||
assert total_messages > MEMORY_WINDOW
|
||||
assert messages_to_process > 0
|
||||
|
||||
expected_consolidate_count = total_messages - KEEP_COUNT
|
||||
assert expected_consolidate_count == 35
|
||||
|
||||
def test_consolidation_skipped_when_within_keep_count(self):
|
||||
"""Test consolidation skipped when total messages <= keep_count."""
|
||||
session = create_session_with_messages("test:skip", 20)
|
||||
|
||||
total_messages = len(session.messages)
|
||||
assert total_messages <= KEEP_COUNT
|
||||
|
||||
old_messages = get_old_messages(session, session.last_consolidated, KEEP_COUNT)
|
||||
assert len(old_messages) == 0
|
||||
|
||||
def test_consolidation_skipped_when_no_new_messages(self):
|
||||
"""Test consolidation skipped when messages_to_process <= 0."""
|
||||
session = create_session_with_messages("test:already_consolidated", 40)
|
||||
session.last_consolidated = len(session.messages) - KEEP_COUNT # 15
|
||||
|
||||
# Add a few more messages
|
||||
for i in range(40, 42):
|
||||
session.add_message("user", f"msg{i}")
|
||||
|
||||
total_messages = len(session.messages)
|
||||
messages_to_process = total_messages - session.last_consolidated
|
||||
assert messages_to_process > 0
|
||||
|
||||
# Simulate last_consolidated catching up
|
||||
session.last_consolidated = total_messages - KEEP_COUNT
|
||||
old_messages = get_old_messages(session, session.last_consolidated, KEEP_COUNT)
|
||||
assert len(old_messages) == 0
|
||||
|
||||
|
||||
class TestLastConsolidatedEdgeCases:
|
||||
"""Test last_consolidated edge cases and data corruption scenarios."""
|
||||
|
||||
def test_last_consolidated_exceeds_message_count(self):
|
||||
"""Test behavior when last_consolidated > len(messages) (data corruption)."""
|
||||
session = create_session_with_messages("test:corruption", 10)
|
||||
session.last_consolidated = 20
|
||||
|
||||
total_messages = len(session.messages)
|
||||
messages_to_process = total_messages - session.last_consolidated
|
||||
assert messages_to_process <= 0
|
||||
|
||||
old_messages = get_old_messages(session, session.last_consolidated, 5)
|
||||
assert len(old_messages) == 0
|
||||
|
||||
def test_last_consolidated_negative_value(self):
|
||||
"""Test behavior with negative last_consolidated (invalid state)."""
|
||||
session = create_session_with_messages("test:negative", 10)
|
||||
session.last_consolidated = -5
|
||||
|
||||
keep_count = 3
|
||||
old_messages = get_old_messages(session, session.last_consolidated, keep_count)
|
||||
|
||||
# messages[-5:-3] with 10 messages gives indices 5,6
|
||||
assert len(old_messages) == 2
|
||||
assert old_messages[0]["content"] == "msg5"
|
||||
assert old_messages[-1]["content"] == "msg6"
|
||||
|
||||
def test_messages_added_after_consolidation(self):
|
||||
"""Test correct behavior when new messages arrive after consolidation."""
|
||||
session = create_session_with_messages("test:new_messages", 40)
|
||||
session.last_consolidated = len(session.messages) - KEEP_COUNT # 15
|
||||
|
||||
# Add new messages after consolidation
|
||||
for i in range(40, 50):
|
||||
session.add_message("user", f"msg{i}")
|
||||
|
||||
total_messages = len(session.messages)
|
||||
old_messages = get_old_messages(session, session.last_consolidated, KEEP_COUNT)
|
||||
expected_consolidate_count = total_messages - KEEP_COUNT - session.last_consolidated
|
||||
|
||||
assert len(old_messages) == expected_consolidate_count
|
||||
assert_messages_content(old_messages, 15, 24)
|
||||
|
||||
def test_slice_behavior_when_indices_overlap(self):
|
||||
"""Test slice behavior when last_consolidated >= total - keep_count."""
|
||||
session = create_session_with_messages("test:overlap", 30)
|
||||
session.last_consolidated = 12
|
||||
|
||||
old_messages = get_old_messages(session, session.last_consolidated, 20)
|
||||
assert len(old_messages) == 0
|
||||
|
||||
|
||||
class TestArchiveAllMode:
|
||||
"""Test archive_all mode (used by /new command)."""
|
||||
|
||||
def test_archive_all_consolidates_everything(self):
|
||||
"""Test archive_all=True consolidates all messages."""
|
||||
session = create_session_with_messages("test:archive_all", 50)
|
||||
|
||||
archive_all = True
|
||||
if archive_all:
|
||||
old_messages = session.messages
|
||||
assert len(old_messages) == 50
|
||||
|
||||
assert session.last_consolidated == 0
|
||||
|
||||
def test_archive_all_resets_last_consolidated(self):
|
||||
"""Test that archive_all mode resets last_consolidated to 0."""
|
||||
session = create_session_with_messages("test:reset", 40)
|
||||
session.last_consolidated = 15
|
||||
|
||||
archive_all = True
|
||||
if archive_all:
|
||||
session.last_consolidated = 0
|
||||
|
||||
assert session.last_consolidated == 0
|
||||
assert len(session.messages) == 40
|
||||
|
||||
def test_archive_all_vs_normal_consolidation(self):
|
||||
"""Test difference between archive_all and normal consolidation."""
|
||||
# Normal consolidation
|
||||
session1 = create_session_with_messages("test:normal", 60)
|
||||
session1.last_consolidated = len(session1.messages) - KEEP_COUNT
|
||||
|
||||
# archive_all mode
|
||||
session2 = create_session_with_messages("test:all", 60)
|
||||
session2.last_consolidated = 0
|
||||
|
||||
assert session1.last_consolidated == 35
|
||||
assert len(session1.messages) == 60
|
||||
assert session2.last_consolidated == 0
|
||||
assert len(session2.messages) == 60
|
||||
|
||||
|
||||
class TestCacheImmutability:
|
||||
"""Test that consolidation doesn't modify session.messages (cache safety)."""
|
||||
|
||||
def test_consolidation_does_not_modify_messages_list(self):
|
||||
"""Test that consolidation leaves messages list unchanged."""
|
||||
session = create_session_with_messages("test:immutable", 50)
|
||||
|
||||
original_messages = session.messages.copy()
|
||||
original_len = len(session.messages)
|
||||
session.last_consolidated = original_len - KEEP_COUNT
|
||||
|
||||
assert len(session.messages) == original_len
|
||||
assert session.messages == original_messages
|
||||
|
||||
def test_get_history_does_not_modify_messages(self):
|
||||
"""Test that get_history doesn't modify messages list."""
|
||||
session = create_session_with_messages("test:history_immutable", 40)
|
||||
original_messages = [m.copy() for m in session.messages]
|
||||
|
||||
for _ in range(5):
|
||||
history = session.get_history(max_messages=10)
|
||||
assert len(history) == 10
|
||||
|
||||
assert len(session.messages) == 40
|
||||
for i, msg in enumerate(session.messages):
|
||||
assert msg["content"] == original_messages[i]["content"]
|
||||
|
||||
def test_consolidation_only_updates_last_consolidated(self):
|
||||
"""Test that consolidation only updates last_consolidated field."""
|
||||
session = create_session_with_messages("test:field_only", 60)
|
||||
|
||||
original_messages = session.messages.copy()
|
||||
original_key = session.key
|
||||
original_metadata = session.metadata.copy()
|
||||
|
||||
session.last_consolidated = len(session.messages) - KEEP_COUNT
|
||||
|
||||
assert session.messages == original_messages
|
||||
assert session.key == original_key
|
||||
assert session.metadata == original_metadata
|
||||
assert session.last_consolidated == 35
|
||||
|
||||
|
||||
class TestSliceLogic:
|
||||
"""Test the slice logic: messages[last_consolidated:-keep_count]."""
|
||||
|
||||
def test_slice_extracts_correct_range(self):
|
||||
"""Test that slice extracts the correct message range."""
|
||||
session = create_session_with_messages("test:slice", 60)
|
||||
|
||||
old_messages = get_old_messages(session, 0, KEEP_COUNT)
|
||||
|
||||
assert len(old_messages) == 35
|
||||
assert_messages_content(old_messages, 0, 34)
|
||||
|
||||
remaining = session.messages[-KEEP_COUNT:]
|
||||
assert len(remaining) == 25
|
||||
assert_messages_content(remaining, 35, 59)
|
||||
|
||||
def test_slice_with_partial_consolidation(self):
|
||||
"""Test slice when some messages already consolidated."""
|
||||
session = create_session_with_messages("test:partial", 70)
|
||||
|
||||
last_consolidated = 30
|
||||
old_messages = get_old_messages(session, last_consolidated, KEEP_COUNT)
|
||||
|
||||
assert len(old_messages) == 15
|
||||
assert_messages_content(old_messages, 30, 44)
|
||||
|
||||
def test_slice_with_various_keep_counts(self):
|
||||
"""Test slice behavior with different keep_count values."""
|
||||
session = create_session_with_messages("test:keep_counts", 50)
|
||||
|
||||
test_cases = [(10, 40), (20, 30), (30, 20), (40, 10)]
|
||||
|
||||
for keep_count, expected_count in test_cases:
|
||||
old_messages = session.messages[0:-keep_count]
|
||||
assert len(old_messages) == expected_count
|
||||
|
||||
def test_slice_when_keep_count_exceeds_messages(self):
|
||||
"""Test slice when keep_count > len(messages)."""
|
||||
session = create_session_with_messages("test:exceed", 10)
|
||||
|
||||
old_messages = session.messages[0:-20]
|
||||
assert len(old_messages) == 0
|
||||
|
||||
|
||||
class TestEmptyAndBoundarySessions:
|
||||
"""Test empty sessions and boundary conditions."""
|
||||
|
||||
def test_empty_session_consolidation(self):
|
||||
"""Test consolidation behavior with empty session."""
|
||||
session = Session(key="test:empty")
|
||||
|
||||
assert len(session.messages) == 0
|
||||
assert session.last_consolidated == 0
|
||||
|
||||
messages_to_process = len(session.messages) - session.last_consolidated
|
||||
assert messages_to_process == 0
|
||||
|
||||
old_messages = get_old_messages(session, session.last_consolidated, KEEP_COUNT)
|
||||
assert len(old_messages) == 0
|
||||
|
||||
def test_single_message_session(self):
|
||||
"""Test consolidation with single message."""
|
||||
session = Session(key="test:single")
|
||||
session.add_message("user", "only message")
|
||||
|
||||
assert len(session.messages) == 1
|
||||
|
||||
old_messages = get_old_messages(session, session.last_consolidated, KEEP_COUNT)
|
||||
assert len(old_messages) == 0
|
||||
|
||||
def test_exactly_keep_count_messages(self):
|
||||
"""Test session with exactly keep_count messages."""
|
||||
session = create_session_with_messages("test:exact", KEEP_COUNT)
|
||||
|
||||
assert len(session.messages) == KEEP_COUNT
|
||||
|
||||
old_messages = get_old_messages(session, session.last_consolidated, KEEP_COUNT)
|
||||
assert len(old_messages) == 0
|
||||
|
||||
def test_just_over_keep_count(self):
|
||||
"""Test session with one message over keep_count."""
|
||||
session = create_session_with_messages("test:over", KEEP_COUNT + 1)
|
||||
|
||||
assert len(session.messages) == 26
|
||||
|
||||
old_messages = get_old_messages(session, session.last_consolidated, KEEP_COUNT)
|
||||
assert len(old_messages) == 1
|
||||
assert old_messages[0]["content"] == "msg0"
|
||||
|
||||
def test_very_large_session(self):
|
||||
"""Test consolidation with very large message count."""
|
||||
session = create_session_with_messages("test:large", 1000)
|
||||
|
||||
assert len(session.messages) == 1000
|
||||
|
||||
old_messages = get_old_messages(session, session.last_consolidated, KEEP_COUNT)
|
||||
assert len(old_messages) == 975
|
||||
assert_messages_content(old_messages, 0, 974)
|
||||
|
||||
remaining = session.messages[-KEEP_COUNT:]
|
||||
assert len(remaining) == 25
|
||||
assert_messages_content(remaining, 975, 999)
|
||||
|
||||
def test_session_with_gaps_in_consolidation(self):
|
||||
"""Test session with potential gaps in consolidation history."""
|
||||
session = create_session_with_messages("test:gaps", 50)
|
||||
session.last_consolidated = 10
|
||||
|
||||
# Add more messages
|
||||
for i in range(50, 60):
|
||||
session.add_message("user", f"msg{i}")
|
||||
|
||||
old_messages = get_old_messages(session, session.last_consolidated, KEEP_COUNT)
|
||||
|
||||
expected_count = 60 - KEEP_COUNT - 10
|
||||
assert len(old_messages) == expected_count
|
||||
assert_messages_content(old_messages, 10, 34)
|
||||
|
||||
|
||||
class TestConsolidationDeduplicationGuard:
|
||||
"""Test that consolidation tasks are deduplicated and serialized."""
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_consolidation_guard_prevents_duplicate_tasks(self, tmp_path: Path) -> None:
|
||||
"""Concurrent messages above memory_window spawn only one consolidation task."""
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.events import InboundMessage
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.providers.base import LLMResponse
|
||||
|
||||
bus = MessageBus()
|
||||
provider = MagicMock()
|
||||
provider.get_default_model.return_value = "test-model"
|
||||
loop = AgentLoop(
|
||||
bus=bus, provider=provider, workspace=tmp_path, model="test-model", memory_window=10
|
||||
)
|
||||
|
||||
loop.provider.chat = AsyncMock(return_value=LLMResponse(content="ok", tool_calls=[]))
|
||||
loop.tools.get_definitions = MagicMock(return_value=[])
|
||||
|
||||
session = loop.sessions.get_or_create("cli:test")
|
||||
for i in range(15):
|
||||
session.add_message("user", f"msg{i}")
|
||||
session.add_message("assistant", f"resp{i}")
|
||||
loop.sessions.save(session)
|
||||
|
||||
consolidation_calls = 0
|
||||
|
||||
async def _fake_consolidate(_session, archive_all: bool = False) -> None:
|
||||
nonlocal consolidation_calls
|
||||
consolidation_calls += 1
|
||||
await asyncio.sleep(0.05)
|
||||
|
||||
loop._consolidate_memory = _fake_consolidate # type: ignore[method-assign]
|
||||
|
||||
msg = InboundMessage(channel="cli", sender_id="user", chat_id="test", content="hello")
|
||||
await loop._process_message(msg)
|
||||
await loop._process_message(msg)
|
||||
await asyncio.sleep(0.1)
|
||||
|
||||
assert consolidation_calls == 1, (
|
||||
f"Expected exactly 1 consolidation, got {consolidation_calls}"
|
||||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_new_command_guard_prevents_concurrent_consolidation(
|
||||
self, tmp_path: Path
|
||||
) -> None:
|
||||
"""/new command does not run consolidation concurrently with in-flight consolidation."""
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.events import InboundMessage
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.providers.base import LLMResponse
|
||||
|
||||
bus = MessageBus()
|
||||
provider = MagicMock()
|
||||
provider.get_default_model.return_value = "test-model"
|
||||
loop = AgentLoop(
|
||||
bus=bus, provider=provider, workspace=tmp_path, model="test-model", memory_window=10
|
||||
)
|
||||
|
||||
loop.provider.chat = AsyncMock(return_value=LLMResponse(content="ok", tool_calls=[]))
|
||||
loop.tools.get_definitions = MagicMock(return_value=[])
|
||||
|
||||
session = loop.sessions.get_or_create("cli:test")
|
||||
for i in range(15):
|
||||
session.add_message("user", f"msg{i}")
|
||||
session.add_message("assistant", f"resp{i}")
|
||||
loop.sessions.save(session)
|
||||
|
||||
consolidation_calls = 0
|
||||
active = 0
|
||||
max_active = 0
|
||||
|
||||
async def _fake_consolidate(_session, archive_all: bool = False) -> None:
|
||||
nonlocal consolidation_calls, active, max_active
|
||||
consolidation_calls += 1
|
||||
active += 1
|
||||
max_active = max(max_active, active)
|
||||
await asyncio.sleep(0.05)
|
||||
active -= 1
|
||||
|
||||
loop._consolidate_memory = _fake_consolidate # type: ignore[method-assign]
|
||||
|
||||
msg = InboundMessage(channel="cli", sender_id="user", chat_id="test", content="hello")
|
||||
await loop._process_message(msg)
|
||||
|
||||
new_msg = InboundMessage(channel="cli", sender_id="user", chat_id="test", content="/new")
|
||||
await loop._process_message(new_msg)
|
||||
await asyncio.sleep(0.1)
|
||||
|
||||
assert consolidation_calls == 2, (
|
||||
f"Expected normal + /new consolidations, got {consolidation_calls}"
|
||||
)
|
||||
assert max_active == 1, (
|
||||
f"Expected serialized consolidation, observed concurrency={max_active}"
|
||||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_consolidation_tasks_are_referenced(self, tmp_path: Path) -> None:
|
||||
"""create_task results are tracked in _consolidation_tasks while in flight."""
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.events import InboundMessage
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.providers.base import LLMResponse
|
||||
|
||||
bus = MessageBus()
|
||||
provider = MagicMock()
|
||||
provider.get_default_model.return_value = "test-model"
|
||||
loop = AgentLoop(
|
||||
bus=bus, provider=provider, workspace=tmp_path, model="test-model", memory_window=10
|
||||
)
|
||||
|
||||
loop.provider.chat = AsyncMock(return_value=LLMResponse(content="ok", tool_calls=[]))
|
||||
loop.tools.get_definitions = MagicMock(return_value=[])
|
||||
|
||||
session = loop.sessions.get_or_create("cli:test")
|
||||
for i in range(15):
|
||||
session.add_message("user", f"msg{i}")
|
||||
session.add_message("assistant", f"resp{i}")
|
||||
loop.sessions.save(session)
|
||||
|
||||
started = asyncio.Event()
|
||||
|
||||
async def _slow_consolidate(_session, archive_all: bool = False) -> None:
|
||||
started.set()
|
||||
await asyncio.sleep(0.1)
|
||||
|
||||
loop._consolidate_memory = _slow_consolidate # type: ignore[method-assign]
|
||||
|
||||
msg = InboundMessage(channel="cli", sender_id="user", chat_id="test", content="hello")
|
||||
await loop._process_message(msg)
|
||||
|
||||
await started.wait()
|
||||
assert len(loop._consolidation_tasks) == 1, "Task must be referenced while in-flight"
|
||||
|
||||
await asyncio.sleep(0.15)
|
||||
assert len(loop._consolidation_tasks) == 0, (
|
||||
"Task reference must be removed after completion"
|
||||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_new_waits_for_inflight_consolidation_and_preserves_messages(
|
||||
self, tmp_path: Path
|
||||
) -> None:
|
||||
"""/new waits for in-flight consolidation and archives before clear."""
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.events import InboundMessage
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.providers.base import LLMResponse
|
||||
|
||||
bus = MessageBus()
|
||||
provider = MagicMock()
|
||||
provider.get_default_model.return_value = "test-model"
|
||||
loop = AgentLoop(
|
||||
bus=bus, provider=provider, workspace=tmp_path, model="test-model", memory_window=10
|
||||
)
|
||||
|
||||
loop.provider.chat = AsyncMock(return_value=LLMResponse(content="ok", tool_calls=[]))
|
||||
loop.tools.get_definitions = MagicMock(return_value=[])
|
||||
|
||||
session = loop.sessions.get_or_create("cli:test")
|
||||
for i in range(15):
|
||||
session.add_message("user", f"msg{i}")
|
||||
session.add_message("assistant", f"resp{i}")
|
||||
loop.sessions.save(session)
|
||||
|
||||
started = asyncio.Event()
|
||||
release = asyncio.Event()
|
||||
archived_count = 0
|
||||
|
||||
async def _fake_consolidate(sess, archive_all: bool = False) -> bool:
|
||||
nonlocal archived_count
|
||||
if archive_all:
|
||||
archived_count = len(sess.messages)
|
||||
return True
|
||||
started.set()
|
||||
await release.wait()
|
||||
return True
|
||||
|
||||
loop._consolidate_memory = _fake_consolidate # type: ignore[method-assign]
|
||||
|
||||
msg = InboundMessage(channel="cli", sender_id="user", chat_id="test", content="hello")
|
||||
await loop._process_message(msg)
|
||||
await started.wait()
|
||||
|
||||
new_msg = InboundMessage(channel="cli", sender_id="user", chat_id="test", content="/new")
|
||||
pending_new = asyncio.create_task(loop._process_message(new_msg))
|
||||
|
||||
await asyncio.sleep(0.02)
|
||||
assert not pending_new.done(), "/new should wait while consolidation is in-flight"
|
||||
|
||||
release.set()
|
||||
response = await pending_new
|
||||
assert response is not None
|
||||
assert "new session started" in response.content.lower()
|
||||
assert archived_count > 0, "Expected /new archival to process a non-empty snapshot"
|
||||
|
||||
session_after = loop.sessions.get_or_create("cli:test")
|
||||
assert session_after.messages == [], "Session should be cleared after successful archival"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_new_does_not_clear_session_when_archive_fails(self, tmp_path: Path) -> None:
|
||||
"""/new must keep session data if archive step reports failure."""
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.events import InboundMessage
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.providers.base import LLMResponse
|
||||
|
||||
bus = MessageBus()
|
||||
provider = MagicMock()
|
||||
provider.get_default_model.return_value = "test-model"
|
||||
loop = AgentLoop(
|
||||
bus=bus, provider=provider, workspace=tmp_path, model="test-model", memory_window=10
|
||||
)
|
||||
|
||||
loop.provider.chat = AsyncMock(return_value=LLMResponse(content="ok", tool_calls=[]))
|
||||
loop.tools.get_definitions = MagicMock(return_value=[])
|
||||
|
||||
session = loop.sessions.get_or_create("cli:test")
|
||||
for i in range(5):
|
||||
session.add_message("user", f"msg{i}")
|
||||
session.add_message("assistant", f"resp{i}")
|
||||
loop.sessions.save(session)
|
||||
before_count = len(session.messages)
|
||||
|
||||
async def _failing_consolidate(sess, archive_all: bool = False) -> bool:
|
||||
if archive_all:
|
||||
return False
|
||||
return True
|
||||
|
||||
loop._consolidate_memory = _failing_consolidate # type: ignore[method-assign]
|
||||
|
||||
new_msg = InboundMessage(channel="cli", sender_id="user", chat_id="test", content="/new")
|
||||
response = await loop._process_message(new_msg)
|
||||
|
||||
assert response is not None
|
||||
assert "failed" in response.content.lower()
|
||||
session_after = loop.sessions.get_or_create("cli:test")
|
||||
assert len(session_after.messages) == before_count, (
|
||||
"Session must remain intact when /new archival fails"
|
||||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_new_archives_only_unconsolidated_messages_after_inflight_task(
|
||||
self, tmp_path: Path
|
||||
) -> None:
|
||||
"""/new should archive only messages not yet consolidated by prior task."""
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.events import InboundMessage
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.providers.base import LLMResponse
|
||||
|
||||
bus = MessageBus()
|
||||
provider = MagicMock()
|
||||
provider.get_default_model.return_value = "test-model"
|
||||
loop = AgentLoop(
|
||||
bus=bus, provider=provider, workspace=tmp_path, model="test-model", memory_window=10
|
||||
)
|
||||
|
||||
loop.provider.chat = AsyncMock(return_value=LLMResponse(content="ok", tool_calls=[]))
|
||||
loop.tools.get_definitions = MagicMock(return_value=[])
|
||||
|
||||
session = loop.sessions.get_or_create("cli:test")
|
||||
for i in range(15):
|
||||
session.add_message("user", f"msg{i}")
|
||||
session.add_message("assistant", f"resp{i}")
|
||||
loop.sessions.save(session)
|
||||
|
||||
started = asyncio.Event()
|
||||
release = asyncio.Event()
|
||||
archived_count = -1
|
||||
|
||||
async def _fake_consolidate(sess, archive_all: bool = False) -> bool:
|
||||
nonlocal archived_count
|
||||
if archive_all:
|
||||
archived_count = len(sess.messages)
|
||||
return True
|
||||
|
||||
started.set()
|
||||
await release.wait()
|
||||
sess.last_consolidated = len(sess.messages) - 3
|
||||
return True
|
||||
|
||||
loop._consolidate_memory = _fake_consolidate # type: ignore[method-assign]
|
||||
|
||||
msg = InboundMessage(channel="cli", sender_id="user", chat_id="test", content="hello")
|
||||
await loop._process_message(msg)
|
||||
await started.wait()
|
||||
|
||||
new_msg = InboundMessage(channel="cli", sender_id="user", chat_id="test", content="/new")
|
||||
pending_new = asyncio.create_task(loop._process_message(new_msg))
|
||||
await asyncio.sleep(0.02)
|
||||
assert not pending_new.done()
|
||||
|
||||
release.set()
|
||||
response = await pending_new
|
||||
|
||||
assert response is not None
|
||||
assert "new session started" in response.content.lower()
|
||||
assert archived_count == 3, (
|
||||
f"Expected only unconsolidated tail to archive, got {archived_count}"
|
||||
)
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_new_cleans_up_consolidation_lock_for_invalidated_session(
|
||||
self, tmp_path: Path
|
||||
) -> None:
|
||||
"""/new should remove lock entry for fully invalidated session key."""
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.events import InboundMessage
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.providers.base import LLMResponse
|
||||
|
||||
bus = MessageBus()
|
||||
provider = MagicMock()
|
||||
provider.get_default_model.return_value = "test-model"
|
||||
loop = AgentLoop(
|
||||
bus=bus, provider=provider, workspace=tmp_path, model="test-model", memory_window=10
|
||||
)
|
||||
|
||||
loop.provider.chat = AsyncMock(return_value=LLMResponse(content="ok", tool_calls=[]))
|
||||
loop.tools.get_definitions = MagicMock(return_value=[])
|
||||
|
||||
session = loop.sessions.get_or_create("cli:test")
|
||||
for i in range(3):
|
||||
session.add_message("user", f"msg{i}")
|
||||
session.add_message("assistant", f"resp{i}")
|
||||
loop.sessions.save(session)
|
||||
|
||||
# Ensure lock exists before /new.
|
||||
loop._consolidation_locks.setdefault(session.key, asyncio.Lock())
|
||||
assert session.key in loop._consolidation_locks
|
||||
|
||||
async def _ok_consolidate(sess, archive_all: bool = False) -> bool:
|
||||
return True
|
||||
|
||||
loop._consolidate_memory = _ok_consolidate # type: ignore[method-assign]
|
||||
|
||||
new_msg = InboundMessage(channel="cli", sender_id="user", chat_id="test", content="/new")
|
||||
response = await loop._process_message(new_msg)
|
||||
|
||||
assert response is not None
|
||||
assert "new session started" in response.content.lower()
|
||||
assert session.key not in loop._consolidation_locks
|
||||
@@ -40,7 +40,7 @@ def test_system_prompt_stays_stable_when_clock_changes(tmp_path, monkeypatch) ->
|
||||
|
||||
|
||||
def test_runtime_context_is_separate_untrusted_user_message(tmp_path) -> None:
|
||||
"""Runtime metadata should be a separate user message before the actual user message."""
|
||||
"""Runtime metadata should be included in the system prompt."""
|
||||
workspace = _make_workspace(tmp_path)
|
||||
builder = ContextBuilder(workspace)
|
||||
|
||||
@@ -51,16 +51,12 @@ def test_runtime_context_is_separate_untrusted_user_message(tmp_path) -> None:
|
||||
chat_id="direct",
|
||||
)
|
||||
|
||||
# Runtime context should be in the system prompt
|
||||
assert messages[0]["role"] == "system"
|
||||
assert "## Current Session" not in messages[0]["content"]
|
||||
|
||||
assert messages[-2]["role"] == "user"
|
||||
runtime_content = messages[-2]["content"]
|
||||
assert isinstance(runtime_content, str)
|
||||
assert ContextBuilder._RUNTIME_CONTEXT_TAG in runtime_content
|
||||
assert "Current Time:" in runtime_content
|
||||
assert "Channel: cli" in runtime_content
|
||||
assert "Chat ID: direct" in runtime_content
|
||||
assert "## Current Session" in messages[0]["content"]
|
||||
assert "Channel: cli" in messages[0]["content"]
|
||||
assert "Chat ID: direct" in messages[0]["content"]
|
||||
|
||||
# The actual user message should be the last message
|
||||
assert messages[-1]["role"] == "user"
|
||||
assert messages[-1]["content"] == "Return exactly: OK"
|
||||
|
||||
@@ -0,0 +1,116 @@
|
||||
"""Tests for EditTool20250728."""
|
||||
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
from nanobot.agent.tools.anthropic.edit import EditTool20250728
|
||||
from nanobot.agent.tools.anthropic.base import CLIResult
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def edit_tool():
|
||||
"""Create an EditTool20250728 instance."""
|
||||
return EditTool20250728()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def temp_file(tmp_path):
|
||||
"""Create a temporary file with some content."""
|
||||
file_path = tmp_path / "test.txt"
|
||||
file_path.write_text("line 1\nline 2\nline 3\n")
|
||||
return file_path
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_view_command(edit_tool, temp_file):
|
||||
"""Test viewing a file with line numbers."""
|
||||
result = await edit_tool(
|
||||
command="view",
|
||||
path=str(temp_file)
|
||||
)
|
||||
assert result.output is not None
|
||||
assert "1|line 1" in result.output
|
||||
assert "2|line 2" in result.output
|
||||
assert "3|line 3" in result.output
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_create_command(edit_tool, tmp_path):
|
||||
"""Test creating a new file."""
|
||||
new_file = tmp_path / "new.txt"
|
||||
result = await edit_tool(
|
||||
command="create",
|
||||
path=str(new_file),
|
||||
file_text="Hello\nWorld\n"
|
||||
)
|
||||
assert result.exit_code == 0
|
||||
assert new_file.exists()
|
||||
assert new_file.read_text() == "Hello\nWorld\n"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_str_replace_command(edit_tool, temp_file):
|
||||
"""Test replacing a unique string."""
|
||||
result = await edit_tool(
|
||||
command="str_replace",
|
||||
path=str(temp_file),
|
||||
old_str="line 2",
|
||||
new_str="LINE TWO"
|
||||
)
|
||||
assert result.exit_code == 0
|
||||
content = temp_file.read_text()
|
||||
assert "LINE TWO" in content
|
||||
assert "line 2" not in content
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_str_replace_non_unique(edit_tool, temp_file):
|
||||
"""Test that str_replace fails on non-unique match."""
|
||||
# Write content with duplicate "line"
|
||||
temp_file.write_text("line 1\nline 2\nline 3\n")
|
||||
result = await edit_tool(
|
||||
command="str_replace",
|
||||
path=str(temp_file),
|
||||
old_str="line", # This appears 3 times
|
||||
new_str="LINE"
|
||||
)
|
||||
assert result.exit_code == 1
|
||||
assert "must match exactly once" in result.error.lower()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_insert_command(edit_tool, temp_file):
|
||||
"""Test inserting text at a specific line."""
|
||||
result = await edit_tool(
|
||||
command="insert",
|
||||
path=str(temp_file),
|
||||
insert_line=1,
|
||||
new_str="inserted line\n"
|
||||
)
|
||||
assert result.exit_code == 0
|
||||
content = temp_file.read_text()
|
||||
lines = content.splitlines()
|
||||
assert lines[1] == "inserted line"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_edit_tool_requires_absolute_path():
|
||||
"""Test edit tool rejects relative paths."""
|
||||
tool = EditTool20250728()
|
||||
|
||||
result = await tool(
|
||||
command="view",
|
||||
path="relative/path.txt"
|
||||
)
|
||||
|
||||
assert isinstance(result, CLIResult)
|
||||
assert result.exit_code == 1
|
||||
assert "absolute" in result.error.lower()
|
||||
|
||||
|
||||
def test_edit_tool_to_params():
|
||||
"""Test edit tool returns correct params."""
|
||||
tool = EditTool20250728()
|
||||
params = tool.to_params()
|
||||
|
||||
assert params["type"] == "text_editor_20250728"
|
||||
assert params["name"] == "str_replace_based_edit_tool"
|
||||
@@ -0,0 +1,95 @@
|
||||
# tests/test_heartbeat_idle.py
|
||||
from datetime import datetime, timedelta
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from nanobot.heartbeat.service import HeartbeatService
|
||||
from nanobot.session.manager import SessionManager
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_heartbeat_skips_when_user_active():
|
||||
"""Test that heartbeat doesn't trigger if user messaged recently."""
|
||||
workspace = Path("/tmp/test-heartbeat")
|
||||
workspace.mkdir(exist_ok=True)
|
||||
|
||||
# Create session with recent user message
|
||||
sessions = SessionManager(workspace)
|
||||
session = sessions.get_or_create("telegram:239824268")
|
||||
session.add_message("user", "Recent message")
|
||||
sessions.save(session)
|
||||
|
||||
# Create heartbeat callback
|
||||
callback_called = False
|
||||
async def on_heartbeat(prompt, metadata=None):
|
||||
nonlocal callback_called
|
||||
callback_called = True
|
||||
return "response"
|
||||
|
||||
# Create heartbeat service
|
||||
service = HeartbeatService(
|
||||
workspace=workspace,
|
||||
on_heartbeat=on_heartbeat,
|
||||
interval_s=1, # Short interval for testing
|
||||
enabled=True,
|
||||
session_manager=sessions,
|
||||
target_session_key="telegram:239824268"
|
||||
)
|
||||
|
||||
# Trigger heartbeat
|
||||
await service._tick()
|
||||
|
||||
# Callback should NOT have been called (user was active recently)
|
||||
assert not callback_called
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_heartbeat_triggers_when_user_idle():
|
||||
"""Test that heartbeat triggers after 30min of inactivity."""
|
||||
workspace = Path("/tmp/test-heartbeat")
|
||||
workspace.mkdir(exist_ok=True)
|
||||
|
||||
# Create HEARTBEAT.md with content
|
||||
heartbeat_file = workspace / "HEARTBEAT.md"
|
||||
heartbeat_file.write_text("# Tasks\n- Check something\n")
|
||||
|
||||
# Create session with old user message (>30min ago)
|
||||
sessions = SessionManager(workspace)
|
||||
session = sessions.get_or_create("telegram:239824268")
|
||||
|
||||
# Manually set old timestamp
|
||||
old_timestamp = (datetime.now() - timedelta(minutes=31)).isoformat()
|
||||
session.messages.append({
|
||||
"role": "user",
|
||||
"content": "Old message",
|
||||
"timestamp": old_timestamp
|
||||
})
|
||||
sessions.save(session)
|
||||
|
||||
# Create heartbeat callback
|
||||
callback_called = False
|
||||
callback_metadata = None
|
||||
|
||||
async def on_heartbeat(prompt, metadata=None):
|
||||
nonlocal callback_called, callback_metadata
|
||||
callback_called = True
|
||||
callback_metadata = metadata
|
||||
return "response"
|
||||
|
||||
# Create heartbeat service
|
||||
service = HeartbeatService(
|
||||
workspace=workspace,
|
||||
on_heartbeat=on_heartbeat,
|
||||
interval_s=1,
|
||||
enabled=True,
|
||||
session_manager=sessions,
|
||||
target_session_key="telegram:239824268"
|
||||
)
|
||||
|
||||
# Trigger heartbeat
|
||||
await service._tick()
|
||||
|
||||
# Callback SHOULD have been called (user idle for >30min)
|
||||
assert callback_called
|
||||
assert callback_metadata == {"suppress_output": True}
|
||||
@@ -0,0 +1,146 @@
|
||||
# tests/test_heartbeat_idle_detection.py
|
||||
"""Tests for heartbeat idle detection with sender_id filtering."""
|
||||
import pytest
|
||||
from datetime import datetime, timedelta
|
||||
from pathlib import Path
|
||||
from nanobot.heartbeat.service import HeartbeatService
|
||||
from nanobot.session.manager import Session, SessionManager
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_idle_detection_ignores_system_messages():
|
||||
"""Test that heartbeat only counts real user messages for idle detection."""
|
||||
# Create session manager and session
|
||||
session_manager = SessionManager(workspace=Path("/tmp/test-heartbeat"))
|
||||
session = session_manager.get_or_create("telegram:239824268")
|
||||
|
||||
# Add a real user message 45 minutes ago
|
||||
real_user_time = datetime.now() - timedelta(minutes=45)
|
||||
session.add_message(
|
||||
"user",
|
||||
"This is a real user message",
|
||||
sender_id="239824268|testuser",
|
||||
timestamp=real_user_time.isoformat()
|
||||
)
|
||||
|
||||
# Add a heartbeat system message 10 minutes ago (should be ignored)
|
||||
heartbeat_time = datetime.now() - timedelta(minutes=10)
|
||||
session.add_message(
|
||||
"user",
|
||||
"Read HEARTBEAT.md...",
|
||||
sender_id="user",
|
||||
timestamp=heartbeat_time.isoformat()
|
||||
)
|
||||
|
||||
# Add an assistant response
|
||||
session.add_message("assistant", "Response to heartbeat")
|
||||
|
||||
# Create heartbeat service with 30 minute idle threshold
|
||||
heartbeat = HeartbeatService(
|
||||
workspace=Path("/tmp/test-heartbeat"),
|
||||
session_manager=session_manager,
|
||||
target_session_key="telegram:239824268",
|
||||
idle_threshold_s=30 * 60, # 30 minutes
|
||||
interval_s=30 * 60,
|
||||
enabled=False # Don't actually start the loop
|
||||
)
|
||||
|
||||
# Manually check idle logic (replicate _tick logic)
|
||||
session = session_manager.get_or_create("telegram:239824268")
|
||||
|
||||
# Find last user message timestamp (should find the 45-minute-old message, not the 10-minute-old one)
|
||||
last_user_timestamp = None
|
||||
for msg in reversed(session.messages):
|
||||
if msg.get("role") == "user":
|
||||
sender_id = msg.get("sender_id")
|
||||
if sender_id == "user":
|
||||
continue
|
||||
last_user_timestamp = msg.get("timestamp")
|
||||
break
|
||||
|
||||
assert last_user_timestamp is not None
|
||||
last_dt = datetime.fromisoformat(last_user_timestamp)
|
||||
elapsed = (datetime.now() - last_dt).total_seconds()
|
||||
|
||||
# Should detect user is idle (45 minutes > 30 minute threshold)
|
||||
assert elapsed >= 30 * 60, f"Expected idle (45min), but elapsed={elapsed/60:.1f}min"
|
||||
# Should NOT be 10 minutes (heartbeat message was ignored)
|
||||
assert elapsed >= 40 * 60, f"Heartbeat message was not ignored, elapsed={elapsed/60:.1f}min"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_idle_detection_counts_real_user_messages():
|
||||
"""Test that heartbeat correctly identifies when user is active."""
|
||||
# Create session manager and session
|
||||
session_manager = SessionManager(workspace=Path("/tmp/test-heartbeat"))
|
||||
session = session_manager.get_or_create("telegram:239824268")
|
||||
|
||||
# Add a real user message 10 minutes ago (recent activity)
|
||||
real_user_time = datetime.now() - timedelta(minutes=10)
|
||||
session.add_message(
|
||||
"user",
|
||||
"This is a recent user message",
|
||||
sender_id="239824268|testuser",
|
||||
timestamp=real_user_time.isoformat()
|
||||
)
|
||||
|
||||
# Create heartbeat service with 30 minute idle threshold
|
||||
heartbeat = HeartbeatService(
|
||||
workspace=Path("/tmp/test-heartbeat"),
|
||||
session_manager=session_manager,
|
||||
target_session_key="telegram:239824268",
|
||||
idle_threshold_s=30 * 60, # 30 minutes
|
||||
interval_s=30 * 60,
|
||||
enabled=False
|
||||
)
|
||||
|
||||
# Find last user message timestamp
|
||||
session = session_manager.get_or_create("telegram:239824268")
|
||||
last_user_timestamp = None
|
||||
for msg in reversed(session.messages):
|
||||
if msg.get("role") == "user":
|
||||
sender_id = msg.get("sender_id")
|
||||
if sender_id == "user":
|
||||
continue
|
||||
last_user_timestamp = msg.get("timestamp")
|
||||
break
|
||||
|
||||
assert last_user_timestamp is not None
|
||||
last_dt = datetime.fromisoformat(last_user_timestamp)
|
||||
elapsed = (datetime.now() - last_dt).total_seconds()
|
||||
|
||||
# Should detect user is active (10 minutes < 30 minute threshold)
|
||||
assert elapsed < 30 * 60, f"Expected active (10min), but elapsed={elapsed/60:.1f}min"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_backwards_compat_messages_without_sender_id():
|
||||
"""Test that old messages without sender_id are treated as real user messages."""
|
||||
# Create session manager and session
|
||||
session_manager = SessionManager(workspace=Path("/tmp/test-heartbeat"))
|
||||
session = session_manager.get_or_create("telegram:239824268")
|
||||
|
||||
# Add an old message without sender_id (backwards compat)
|
||||
old_time = datetime.now() - timedelta(minutes=20)
|
||||
session.add_message(
|
||||
"user",
|
||||
"Old message without sender_id",
|
||||
timestamp=old_time.isoformat()
|
||||
)
|
||||
|
||||
# Find last user message timestamp (should find the old message)
|
||||
last_user_timestamp = None
|
||||
for msg in reversed(session.messages):
|
||||
if msg.get("role") == "user":
|
||||
sender_id = msg.get("sender_id")
|
||||
if sender_id == "user":
|
||||
continue
|
||||
last_user_timestamp = msg.get("timestamp")
|
||||
break
|
||||
|
||||
assert last_user_timestamp is not None
|
||||
last_dt = datetime.fromisoformat(last_user_timestamp)
|
||||
elapsed = (datetime.now() - last_dt).total_seconds()
|
||||
|
||||
# Should accept old message (backwards compat)
|
||||
assert elapsed < 25 * 60, f"Old message not counted, elapsed={elapsed/60:.1f}min"
|
||||
@@ -3,27 +3,12 @@ import asyncio
|
||||
import pytest
|
||||
|
||||
from nanobot.heartbeat.service import HeartbeatService
|
||||
from nanobot.providers.base import LLMResponse, ToolCallRequest
|
||||
|
||||
|
||||
class DummyProvider:
|
||||
def __init__(self, responses: list[LLMResponse]):
|
||||
self._responses = list(responses)
|
||||
|
||||
async def chat(self, *args, **kwargs) -> LLMResponse:
|
||||
if self._responses:
|
||||
return self._responses.pop(0)
|
||||
return LLMResponse(content="", tool_calls=[])
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_start_is_idempotent(tmp_path) -> None:
|
||||
provider = DummyProvider([])
|
||||
|
||||
service = HeartbeatService(
|
||||
workspace=tmp_path,
|
||||
provider=provider,
|
||||
model="openai/gpt-4o-mini",
|
||||
interval_s=9999,
|
||||
enabled=True,
|
||||
)
|
||||
@@ -38,80 +23,36 @@ async def test_start_is_idempotent(tmp_path) -> None:
|
||||
await asyncio.sleep(0)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_decide_returns_skip_when_no_tool_call(tmp_path) -> None:
|
||||
provider = DummyProvider([LLMResponse(content="no tool call", tool_calls=[])])
|
||||
service = HeartbeatService(
|
||||
workspace=tmp_path,
|
||||
provider=provider,
|
||||
model="openai/gpt-4o-mini",
|
||||
)
|
||||
|
||||
action, tasks = await service._decide("heartbeat content")
|
||||
assert action == "skip"
|
||||
assert tasks == ""
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_trigger_now_executes_when_decision_is_run(tmp_path) -> None:
|
||||
(tmp_path / "HEARTBEAT.md").write_text("- [ ] do thing", encoding="utf-8")
|
||||
|
||||
provider = DummyProvider([
|
||||
LLMResponse(
|
||||
content="",
|
||||
tool_calls=[
|
||||
ToolCallRequest(
|
||||
id="hb_1",
|
||||
name="heartbeat",
|
||||
arguments={"action": "run", "tasks": "check open tasks"},
|
||||
)
|
||||
],
|
||||
)
|
||||
])
|
||||
called_with: list[tuple[str, dict | None]] = []
|
||||
|
||||
called_with: list[str] = []
|
||||
|
||||
async def _on_execute(tasks: str) -> str:
|
||||
called_with.append(tasks)
|
||||
async def _on_heartbeat(prompt: str, metadata: dict | None = None) -> str:
|
||||
called_with.append((prompt, metadata))
|
||||
return "done"
|
||||
|
||||
service = HeartbeatService(
|
||||
workspace=tmp_path,
|
||||
provider=provider,
|
||||
model="openai/gpt-4o-mini",
|
||||
on_execute=_on_execute,
|
||||
on_heartbeat=_on_heartbeat,
|
||||
)
|
||||
|
||||
result = await service.trigger_now()
|
||||
assert result == "done"
|
||||
assert called_with == ["check open tasks"]
|
||||
assert len(called_with) == 1
|
||||
prompt, metadata = called_with[0]
|
||||
assert "HEARTBEAT.md" in prompt
|
||||
assert metadata == {"suppress_output": True}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_trigger_now_returns_none_when_decision_is_skip(tmp_path) -> None:
|
||||
async def test_trigger_now_returns_none_when_no_callback(tmp_path) -> None:
|
||||
(tmp_path / "HEARTBEAT.md").write_text("- [ ] do thing", encoding="utf-8")
|
||||
|
||||
provider = DummyProvider([
|
||||
LLMResponse(
|
||||
content="",
|
||||
tool_calls=[
|
||||
ToolCallRequest(
|
||||
id="hb_1",
|
||||
name="heartbeat",
|
||||
arguments={"action": "skip"},
|
||||
)
|
||||
],
|
||||
)
|
||||
])
|
||||
|
||||
async def _on_execute(tasks: str) -> str:
|
||||
return tasks
|
||||
|
||||
service = HeartbeatService(
|
||||
workspace=tmp_path,
|
||||
provider=provider,
|
||||
model="openai/gpt-4o-mini",
|
||||
on_execute=_on_execute,
|
||||
on_heartbeat=None, # No callback
|
||||
)
|
||||
|
||||
assert await service.trigger_now() is None
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
"""Tests for the hook channel."""
|
||||
|
||||
import pytest
|
||||
from unittest.mock import MagicMock
|
||||
from nanobot.channels.hook import HookChannel
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.bus.events import OutboundMessage
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def bus():
|
||||
return MessageBus()
|
||||
|
||||
|
||||
def test_hook_channel_name():
|
||||
bus = MessageBus()
|
||||
channel = HookChannel(bus)
|
||||
assert channel.name == "hook"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_hook_channel_send_is_noop():
|
||||
"""send() should not raise and should not do anything."""
|
||||
bus = MessageBus()
|
||||
channel = HookChannel(bus)
|
||||
msg = OutboundMessage(channel="hook", chat_id="test", content="hello")
|
||||
await channel.send(msg) # Should not raise
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_hook_channel_start_stop():
|
||||
bus = MessageBus()
|
||||
channel = HookChannel(bus)
|
||||
await channel.start()
|
||||
assert channel.is_running
|
||||
await channel.stop()
|
||||
assert not channel.is_running
|
||||
@@ -0,0 +1,24 @@
|
||||
"""Tests for HooksConfig with named tokens."""
|
||||
|
||||
from nanobot.config.schema import HooksConfig
|
||||
|
||||
|
||||
def test_hooks_config_named_tokens():
|
||||
"""Named tokens dict should work."""
|
||||
config = HooksConfig(enabled=True, tokens={"gitea": "secret1", "ha": "secret2"})
|
||||
assert config.tokens == {"gitea": "secret1", "ha": "secret2"}
|
||||
|
||||
|
||||
def test_hooks_config_resolve_token():
|
||||
"""resolve_token should return token name for a matching token."""
|
||||
config = HooksConfig(enabled=True, tokens={"gitea": "secret1", "ha": "secret2"})
|
||||
assert config.resolve_token("secret1") == "gitea"
|
||||
assert config.resolve_token("secret2") == "ha"
|
||||
assert config.resolve_token("unknown") is None
|
||||
|
||||
|
||||
def test_hooks_config_has_tokens():
|
||||
"""has_tokens should be True if tokens dict is non-empty."""
|
||||
assert HooksConfig(enabled=True, tokens={"webhook": "secret"}).has_tokens
|
||||
assert not HooksConfig(enabled=True).has_tokens
|
||||
assert not HooksConfig(enabled=True, tokens={}).has_tokens
|
||||
@@ -0,0 +1,128 @@
|
||||
"""End-to-end integration test for hooks → bus → correlation → response."""
|
||||
|
||||
import asyncio
|
||||
import pytest
|
||||
from aiohttp.test_utils import TestClient, TestServer
|
||||
from nanobot.hooks.server import HooksServer
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.bus.events import InboundMessage, OutboundMessage
|
||||
from nanobot.config.schema import HooksConfig
|
||||
from nanobot.channels.hook import HookChannel
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def bus():
|
||||
return MessageBus()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def config():
|
||||
return HooksConfig(
|
||||
enabled=True,
|
||||
tokens={"gitea": "gitea-secret", "ha": "ha-secret"},
|
||||
timeout_seconds=5,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def server(bus, config):
|
||||
return HooksServer(host="127.0.0.1", port=0, config=config, bus=bus)
|
||||
|
||||
|
||||
async def fake_agent_loop(bus: MessageBus):
|
||||
"""Simulate agent loop: consume inbound, process, publish outbound."""
|
||||
msg = await asyncio.wait_for(bus.consume_inbound(), timeout=3.0)
|
||||
response_content = f"Processed: {msg.content}"
|
||||
await bus.publish_outbound(OutboundMessage(
|
||||
channel=msg.channel,
|
||||
chat_id=msg.chat_id,
|
||||
content=response_content,
|
||||
metadata=msg.metadata or {},
|
||||
))
|
||||
|
||||
|
||||
async def fake_dispatch_loop(bus: MessageBus, hook_channel: HookChannel):
|
||||
"""Simulate outbound dispatcher: consume outbound, resolve correlation, dispatch."""
|
||||
msg = await asyncio.wait_for(bus.consume_outbound(), timeout=3.0)
|
||||
bus.resolve_correlation(msg)
|
||||
if msg.channel == "hook":
|
||||
await hook_channel.send(msg)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_full_hook_flow_default_channel(server, bus):
|
||||
"""Hook with default channel: message goes through bus, response returned to HTTP caller."""
|
||||
hook_channel = HookChannel(bus)
|
||||
client = TestClient(TestServer(server._app))
|
||||
async with client:
|
||||
async def do_request():
|
||||
return await client.post(
|
||||
"/hooks",
|
||||
json={"message": "deploy started"},
|
||||
headers={"Authorization": "Bearer gitea-secret"},
|
||||
)
|
||||
|
||||
# Run request + fake agent + fake dispatcher concurrently
|
||||
request_task = asyncio.create_task(do_request())
|
||||
agent_task = asyncio.create_task(fake_agent_loop(bus))
|
||||
dispatch_task = asyncio.create_task(fake_dispatch_loop(bus, hook_channel))
|
||||
|
||||
resp = await asyncio.wait_for(request_task, timeout=5.0)
|
||||
await agent_task
|
||||
await dispatch_task
|
||||
|
||||
assert resp.status == 200
|
||||
data = await resp.json()
|
||||
assert data["ok"] is True
|
||||
assert "deploy started" in data["response"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_full_hook_flow_telegram_channel(server, bus):
|
||||
"""Hook targeting telegram: uses telegram session, response still returned to HTTP caller."""
|
||||
hook_channel = HookChannel(bus)
|
||||
client = TestClient(TestServer(server._app))
|
||||
async with client:
|
||||
async def do_request():
|
||||
return await client.post(
|
||||
"/hooks",
|
||||
json={"message": "doorbell rang", "channel": "telegram", "chat_id": "239824268"},
|
||||
headers={"Authorization": "Bearer ha-secret"},
|
||||
)
|
||||
|
||||
request_task = asyncio.create_task(do_request())
|
||||
agent_task = asyncio.create_task(fake_agent_loop(bus))
|
||||
dispatch_task = asyncio.create_task(fake_dispatch_loop(bus, hook_channel))
|
||||
|
||||
resp = await asyncio.wait_for(request_task, timeout=5.0)
|
||||
await agent_task
|
||||
await dispatch_task
|
||||
|
||||
assert resp.status == 200
|
||||
data = await resp.json()
|
||||
assert data["ok"] is True
|
||||
assert "doorbell rang" in data["response"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_named_token_identification(server, bus):
|
||||
"""Different tokens should produce different hook_source in metadata."""
|
||||
client = TestClient(TestServer(server._app))
|
||||
async with client:
|
||||
asyncio.create_task(client.post(
|
||||
"/hooks",
|
||||
json={"message": "from gitea"},
|
||||
headers={"Authorization": "Bearer gitea-secret"},
|
||||
))
|
||||
msg1 = await asyncio.wait_for(bus.consume_inbound(), timeout=2.0)
|
||||
assert msg1.metadata["hook_source"] == "gitea"
|
||||
assert msg1.chat_id == "gitea"
|
||||
|
||||
asyncio.create_task(client.post(
|
||||
"/hooks",
|
||||
json={"message": "from ha"},
|
||||
headers={"Authorization": "Bearer ha-secret"},
|
||||
))
|
||||
msg2 = await asyncio.wait_for(bus.consume_inbound(), timeout=2.0)
|
||||
assert msg2.metadata["hook_source"] == "ha"
|
||||
assert msg2.chat_id == "ha"
|
||||
@@ -0,0 +1,176 @@
|
||||
"""Tests for the rewritten hooks server."""
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import pytest
|
||||
from aiohttp import web
|
||||
from aiohttp.test_utils import AioHTTPTestCase, unittest_run_loop, TestClient, TestServer
|
||||
from nanobot.hooks.server import HooksServer
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.bus.events import InboundMessage, OutboundMessage
|
||||
from nanobot.config.schema import HooksConfig
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def bus():
|
||||
return MessageBus()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def config():
|
||||
return HooksConfig(
|
||||
enabled=True,
|
||||
tokens={"test-hook": "test-secret-123"},
|
||||
timeout_seconds=5,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def server(bus, config):
|
||||
return HooksServer(host="127.0.0.1", port=0, config=config, bus=bus)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_health_check(server):
|
||||
client = TestClient(TestServer(server._app))
|
||||
async with client:
|
||||
resp = await client.get("/health")
|
||||
assert resp.status == 200
|
||||
data = await resp.json()
|
||||
assert data["status"] == "ok"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_unauthorized_without_token(server):
|
||||
client = TestClient(TestServer(server._app))
|
||||
async with client:
|
||||
resp = await client.post("/hooks", json={"message": "test"})
|
||||
assert resp.status == 401
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_unauthorized_wrong_token(server):
|
||||
client = TestClient(TestServer(server._app))
|
||||
async with client:
|
||||
resp = await client.post(
|
||||
"/hooks",
|
||||
json={"message": "test"},
|
||||
headers={"Authorization": "Bearer wrong-token"},
|
||||
)
|
||||
assert resp.status == 401
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_missing_message_field(server, bus):
|
||||
client = TestClient(TestServer(server._app))
|
||||
async with client:
|
||||
resp = await client.post(
|
||||
"/hooks",
|
||||
json={"not_message": "test"},
|
||||
headers={"Authorization": "Bearer test-secret-123"},
|
||||
)
|
||||
assert resp.status == 400
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_hook_publishes_to_bus(server, bus):
|
||||
"""Hook should publish InboundMessage to bus and the message should contain hook prefix."""
|
||||
client = TestClient(TestServer(server._app))
|
||||
async with client:
|
||||
# Send hook request in background (it will block waiting for correlation)
|
||||
async def send_request():
|
||||
return await client.post(
|
||||
"/hooks",
|
||||
json={"message": "hello from webhook"},
|
||||
headers={"Authorization": "Bearer test-secret-123"},
|
||||
)
|
||||
|
||||
task = asyncio.create_task(send_request())
|
||||
|
||||
# Consume the inbound message
|
||||
msg = await asyncio.wait_for(bus.consume_inbound(), timeout=2.0)
|
||||
|
||||
assert msg.channel == "hook"
|
||||
assert msg.chat_id == "test-hook" # defaults to token name
|
||||
assert msg.metadata.get("hook_source") == "test-hook"
|
||||
assert msg.metadata.get("correlation_id") is not None
|
||||
|
||||
# Simulate agent response by resolving correlation
|
||||
bus.resolve_correlation(OutboundMessage(
|
||||
channel="hook",
|
||||
chat_id="test-hook",
|
||||
content="agent says hi",
|
||||
metadata={"correlation_id": msg.metadata["correlation_id"]},
|
||||
))
|
||||
|
||||
resp = await asyncio.wait_for(task, timeout=2.0)
|
||||
assert resp.status == 200
|
||||
data = await resp.json()
|
||||
assert data["ok"] is True
|
||||
assert data["response"] == "agent says hi"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_hook_with_custom_channel(server, bus):
|
||||
"""Hook targeting telegram should use telegram channel in InboundMessage."""
|
||||
client = TestClient(TestServer(server._app))
|
||||
async with client:
|
||||
async def send_request():
|
||||
return await client.post(
|
||||
"/hooks",
|
||||
json={"message": "notify user", "channel": "telegram", "chat_id": "239824268"},
|
||||
headers={"Authorization": "Bearer test-secret-123"},
|
||||
)
|
||||
|
||||
task = asyncio.create_task(send_request())
|
||||
|
||||
msg = await asyncio.wait_for(bus.consume_inbound(), timeout=2.0)
|
||||
assert msg.channel == "telegram"
|
||||
assert msg.chat_id == "239824268"
|
||||
assert msg.session_key == "telegram:239824268"
|
||||
|
||||
bus.resolve_correlation(OutboundMessage(
|
||||
channel="telegram",
|
||||
chat_id="239824268",
|
||||
content="done",
|
||||
metadata={"correlation_id": msg.metadata["correlation_id"]},
|
||||
))
|
||||
|
||||
resp = await asyncio.wait_for(task, timeout=2.0)
|
||||
assert resp.status == 200
|
||||
data = await resp.json()
|
||||
assert data["response"] == "done"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_hook_timeout_returns_504(bus):
|
||||
"""If agent doesn't respond in time, return 504."""
|
||||
config = HooksConfig(enabled=True, tokens={"test-hook": "test-secret-123"}, timeout_seconds=1)
|
||||
server = HooksServer(host="127.0.0.1", port=0, config=config, bus=bus)
|
||||
client = TestClient(TestServer(server._app))
|
||||
async with client:
|
||||
resp = await client.post(
|
||||
"/hooks",
|
||||
json={"message": "slow request"},
|
||||
headers={"Authorization": "Bearer test-secret-123"},
|
||||
)
|
||||
assert resp.status == 504
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_hook_timeout_zero_returns_202(server, bus):
|
||||
"""timeout=0 should return 202 immediately without waiting."""
|
||||
client = TestClient(TestServer(server._app))
|
||||
async with client:
|
||||
resp = await client.post(
|
||||
"/hooks",
|
||||
json={"message": "fire and forget", "timeout": 0},
|
||||
headers={"Authorization": "Bearer test-secret-123"},
|
||||
)
|
||||
assert resp.status == 202
|
||||
data = await resp.json()
|
||||
assert data["ok"] is True
|
||||
|
||||
# Message should still be on the bus
|
||||
msg = await asyncio.wait_for(bus.consume_inbound(), timeout=1.0)
|
||||
assert msg.content == "fire and forget"
|
||||
@@ -0,0 +1,110 @@
|
||||
# tests/test_idle_heartbeat_integration.py
|
||||
import pytest
|
||||
import asyncio
|
||||
from pathlib import Path
|
||||
from datetime import datetime, timedelta
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.heartbeat.service import HeartbeatService
|
||||
from nanobot.session.manager import SessionManager
|
||||
from nanobot.providers.base import LLMProvider, LLMResponse
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_idle_heartbeat_end_to_end(tmp_path):
|
||||
"""
|
||||
Integration test: heartbeat triggers when idle, runs in main session,
|
||||
output is suppressed, session contains [HIDDEN:signature] content.
|
||||
"""
|
||||
workspace = tmp_path / "test-integration"
|
||||
workspace.mkdir()
|
||||
|
||||
# Use test-specific session key
|
||||
test_session_key = "telegram:test_integration"
|
||||
|
||||
# Create HEARTBEAT.md with content
|
||||
heartbeat_file = workspace / "HEARTBEAT.md"
|
||||
heartbeat_file.write_text("# Test Task\n- Check something")
|
||||
|
||||
# Create mock provider
|
||||
provider = MagicMock(spec=LLMProvider)
|
||||
provider.chat = AsyncMock(return_value=LLMResponse(
|
||||
content="Heartbeat executed successfully",
|
||||
tool_calls=[] # has_tool_calls is a property, not a parameter
|
||||
))
|
||||
provider.get_default_model = MagicMock(return_value="test-model")
|
||||
provider.thinking_budget = 0
|
||||
|
||||
# Create components
|
||||
bus = MessageBus()
|
||||
sessions = SessionManager(workspace)
|
||||
# Override sessions_dir to use tmp_path for test isolation
|
||||
sessions.sessions_dir = tmp_path / "sessions"
|
||||
sessions.sessions_dir.mkdir()
|
||||
loop = AgentLoop(
|
||||
bus=bus,
|
||||
provider=provider,
|
||||
workspace=workspace,
|
||||
session_manager=sessions
|
||||
)
|
||||
|
||||
# Create session with old user message
|
||||
session = sessions.get_or_create(test_session_key)
|
||||
old_timestamp = (datetime.now() - timedelta(minutes=31)).isoformat()
|
||||
session.messages.append({
|
||||
"role": "user",
|
||||
"content": "Old user message",
|
||||
"timestamp": old_timestamp
|
||||
})
|
||||
sessions.save(session)
|
||||
|
||||
# Create heartbeat callback
|
||||
async def on_heartbeat(prompt: str, metadata=None):
|
||||
return await loop.process_direct(
|
||||
prompt,
|
||||
session_key=test_session_key,
|
||||
channel="telegram",
|
||||
chat_id="test_integration",
|
||||
metadata=metadata,
|
||||
)
|
||||
|
||||
# Create heartbeat service
|
||||
heartbeat = HeartbeatService(
|
||||
workspace=workspace,
|
||||
on_heartbeat=on_heartbeat,
|
||||
interval_s=1,
|
||||
enabled=True,
|
||||
session_manager=sessions,
|
||||
target_session_key=test_session_key,
|
||||
idle_threshold_s=30 * 60,
|
||||
)
|
||||
|
||||
# Trigger heartbeat
|
||||
await heartbeat._tick()
|
||||
|
||||
# Reload session from disk
|
||||
sessions._cache.clear() # Clear cache to force reload
|
||||
session = sessions.get_or_create(test_session_key)
|
||||
|
||||
# Verify:
|
||||
# 1. Session has new messages
|
||||
assert len(session.messages) > 1
|
||||
|
||||
# 2. Find the heartbeat response (assistant message with signed visibility marker)
|
||||
heartbeat_messages = [
|
||||
m for m in session.messages
|
||||
if m.get("role") == "assistant" and "[HIDDEN:" in m.get("content", "")
|
||||
]
|
||||
assert len(heartbeat_messages) == 1, "Expected exactly 1 [HIDDEN:signature] heartbeat message"
|
||||
|
||||
# 3. Verify content is prefixed with [HIDDEN:signature]
|
||||
heartbeat_msg = heartbeat_messages[0]
|
||||
assert heartbeat_msg["content"].startswith("[HIDDEN:")
|
||||
# Verify signature format (8-char hex)
|
||||
content = heartbeat_msg["content"]
|
||||
prefix_end = content.index("]")
|
||||
signature = content[8:prefix_end] # Skip "[HIDDEN:" to get signature
|
||||
assert len(signature) == 8, f"Expected 8-char signature, got {len(signature)}"
|
||||
assert all(c in "0123456789abcdef" for c in signature), "Signature should be hex"
|
||||
assert "Heartbeat executed successfully" in heartbeat_msg["content"]
|
||||
@@ -0,0 +1,78 @@
|
||||
"""Test auto-consolidation on long context 429 errors."""
|
||||
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
from nanobot.providers.base import LongContextError, LLMResponse
|
||||
|
||||
|
||||
def test_long_context_error_is_exception():
|
||||
"""LongContextError should be a distinct exception class."""
|
||||
err = LongContextError("too long")
|
||||
assert isinstance(err, Exception)
|
||||
assert str(err) == "too long"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_provider_raises_long_context_error_on_long_context_429():
|
||||
"""Provider should raise LongContextError immediately for long-context 429s."""
|
||||
from nanobot.providers.anthropic_oauth import AnthropicOAuthProvider
|
||||
|
||||
provider = AnthropicOAuthProvider(
|
||||
oauth_token="sk-ant-oat01-test-token",
|
||||
default_model="claude-sonnet-4-6",
|
||||
)
|
||||
|
||||
mock_response = MagicMock()
|
||||
mock_response.status_code = 429
|
||||
mock_response.text = '{"type":"error","error":{"type":"rate_limit_error","message":"Extra usage is required for long context requests."}}'
|
||||
mock_response.headers = {}
|
||||
|
||||
mock_client = AsyncMock()
|
||||
mock_client.post.return_value = mock_response
|
||||
provider._client = mock_client
|
||||
|
||||
with pytest.raises(LongContextError, match="Context too long"):
|
||||
await provider._make_request(
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
)
|
||||
|
||||
# Should NOT retry — only one call
|
||||
assert mock_client.post.call_count == 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_provider_retries_normal_429():
|
||||
"""Provider should still retry normal 429s (not long-context)."""
|
||||
from nanobot.providers.anthropic_oauth import AnthropicOAuthProvider
|
||||
|
||||
provider = AnthropicOAuthProvider(
|
||||
oauth_token="sk-ant-oat01-test-token",
|
||||
default_model="claude-sonnet-4-6",
|
||||
)
|
||||
|
||||
rate_limit_response = MagicMock()
|
||||
rate_limit_response.status_code = 429
|
||||
rate_limit_response.text = '{"type":"error","error":{"type":"rate_limit_error","message":"Rate limit exceeded"}}'
|
||||
rate_limit_response.headers = {}
|
||||
|
||||
success_response = MagicMock()
|
||||
success_response.status_code = 200
|
||||
success_response.headers = {}
|
||||
success_response.json.return_value = {
|
||||
"content": [{"type": "text", "text": "ok"}],
|
||||
"stop_reason": "end_turn",
|
||||
"usage": {"input_tokens": 1, "output_tokens": 1},
|
||||
}
|
||||
|
||||
mock_client = AsyncMock()
|
||||
mock_client.post.side_effect = [rate_limit_response, success_response]
|
||||
provider._client = mock_client
|
||||
|
||||
result = await provider._make_request(
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
)
|
||||
|
||||
# Should have retried and succeeded
|
||||
assert mock_client.post.call_count == 2
|
||||
assert result["stop_reason"] == "end_turn"
|
||||
@@ -676,7 +676,7 @@ async def test_on_media_message_respects_declared_size_limit(
|
||||
assert client.download_calls == []
|
||||
assert len(handled) == 1
|
||||
assert handled[0]["media"] == []
|
||||
assert handled[0]["metadata"]["attachments"] == []
|
||||
assert handled[0]["metadata"].get("attachments", []) == []
|
||||
assert "[attachment: large.bin - too large]" in handled[0]["content"]
|
||||
|
||||
|
||||
@@ -712,7 +712,7 @@ async def test_on_media_message_uses_server_limit_when_smaller_than_local_limit(
|
||||
assert client.download_calls == []
|
||||
assert len(handled) == 1
|
||||
assert handled[0]["media"] == []
|
||||
assert handled[0]["metadata"]["attachments"] == []
|
||||
assert handled[0]["metadata"].get("attachments", []) == []
|
||||
assert "[attachment: large.bin - too large]" in handled[0]["content"]
|
||||
|
||||
|
||||
@@ -746,7 +746,7 @@ async def test_on_media_message_handles_download_error(monkeypatch, tmp_path) ->
|
||||
assert len(client.download_calls) == 1
|
||||
assert len(handled) == 1
|
||||
assert handled[0]["media"] == []
|
||||
assert handled[0]["metadata"]["attachments"] == []
|
||||
assert handled[0]["metadata"].get("attachments", []) == []
|
||||
assert "[attachment: photo.png - download failed]" in handled[0]["content"]
|
||||
|
||||
|
||||
@@ -830,7 +830,7 @@ async def test_on_media_message_handles_decrypt_error(monkeypatch, tmp_path) ->
|
||||
|
||||
assert len(handled) == 1
|
||||
assert handled[0]["media"] == []
|
||||
assert handled[0]["metadata"]["attachments"] == []
|
||||
assert handled[0]["metadata"].get("attachments", []) == []
|
||||
assert "[attachment: secret.txt - download failed]" in handled[0]["content"]
|
||||
|
||||
|
||||
@@ -972,7 +972,6 @@ async def test_send_passes_thread_relates_to_to_attachment_upload(monkeypatch) -
|
||||
captured: dict[str, object] = {}
|
||||
|
||||
async def _fake_upload_and_send_attachment(
|
||||
*,
|
||||
room_id: str,
|
||||
path: Path,
|
||||
limit_bytes: int,
|
||||
|
||||
@@ -0,0 +1,25 @@
|
||||
"""Tests for screenshot media tracking."""
|
||||
|
||||
import pytest
|
||||
import base64
|
||||
from pathlib import Path
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.agent.tools.anthropic.base import ToolResult
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_media_tracking_saves_screenshots():
|
||||
"""Test that screenshots are saved to disk and tracked."""
|
||||
# This is more of an integration test
|
||||
# Test the media saving logic separately
|
||||
|
||||
# Create fake screenshot data
|
||||
fake_png = b"\x89PNG\r\n\x1a\n" # PNG header
|
||||
base64_image = base64.b64encode(fake_png).decode()
|
||||
|
||||
result = ToolResult(base64_image=base64_image)
|
||||
|
||||
# Verify we can decode it
|
||||
decoded = base64.b64decode(result.base64_image)
|
||||
assert decoded == fake_png
|
||||
@@ -0,0 +1,129 @@
|
||||
"""Test mem0 fact extraction calls provider with thinking disabled."""
|
||||
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
from pathlib import Path
|
||||
|
||||
from nanobot.providers.base import LLMResponse
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_provider():
|
||||
provider = AsyncMock()
|
||||
provider.chat = AsyncMock(return_value=LLMResponse(
|
||||
content='{"facts": ["user likes Python", "user works on nanobot"]}',
|
||||
finish_reason="end_turn",
|
||||
))
|
||||
return provider
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mem0_store(tmp_path):
|
||||
"""Create a Mem0MemoryStore with mocked mem0 dependency."""
|
||||
# We can't import Mem0MemoryStore at module level because it requires
|
||||
# the mem0 package. Instead, we test extract_facts as a standalone method
|
||||
# by constructing a minimal instance.
|
||||
try:
|
||||
from nanobot.agent.memory_mem0 import Mem0MemoryStore
|
||||
store = Mem0MemoryStore(workspace=tmp_path)
|
||||
return store
|
||||
except ImportError:
|
||||
pytest.skip("mem0 not installed")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_extract_facts_passes_thinking_budget_zero(mock_provider):
|
||||
"""extract_facts must pass thinking_budget=0 to provider.chat().
|
||||
|
||||
Without this, the provider inherits its instance default (e.g. 10000),
|
||||
causing the model to spend tokens on thinking instead of outputting JSON.
|
||||
"""
|
||||
try:
|
||||
from nanobot.agent.memory_mem0 import Mem0MemoryStore
|
||||
except ImportError:
|
||||
pytest.skip("mem0 not installed")
|
||||
|
||||
# Create a minimal instance without full mem0 init
|
||||
store = object.__new__(Mem0MemoryStore)
|
||||
store.custom_prompt = "Extract facts as JSON: "
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": "I like Python programming"},
|
||||
{"role": "assistant", "content": "That's great! Python is versatile."},
|
||||
]
|
||||
|
||||
facts = await store.extract_facts(messages, mock_provider, "claude-sonnet-4-6")
|
||||
|
||||
# Verify provider.chat was called with thinking_budget=0
|
||||
mock_provider.chat.assert_called_once()
|
||||
call_kwargs = mock_provider.chat.call_args.kwargs
|
||||
assert call_kwargs["thinking_budget"] == 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_extract_facts_returns_parsed_facts(mock_provider):
|
||||
"""extract_facts should parse JSON response into a list of fact strings."""
|
||||
try:
|
||||
from nanobot.agent.memory_mem0 import Mem0MemoryStore
|
||||
except ImportError:
|
||||
pytest.skip("mem0 not installed")
|
||||
|
||||
store = object.__new__(Mem0MemoryStore)
|
||||
store.custom_prompt = "Extract facts as JSON: "
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": "I like Python programming"},
|
||||
]
|
||||
|
||||
facts = await store.extract_facts(messages, mock_provider, "claude-sonnet-4-6")
|
||||
|
||||
assert facts == ["user likes Python", "user works on nanobot"]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_extract_facts_handles_empty_response():
|
||||
"""extract_facts should return empty list when provider returns no content."""
|
||||
try:
|
||||
from nanobot.agent.memory_mem0 import Mem0MemoryStore
|
||||
except ImportError:
|
||||
pytest.skip("mem0 not installed")
|
||||
|
||||
provider = AsyncMock()
|
||||
provider.chat = AsyncMock(return_value=LLMResponse(
|
||||
content="",
|
||||
finish_reason="end_turn",
|
||||
))
|
||||
|
||||
store = object.__new__(Mem0MemoryStore)
|
||||
store.custom_prompt = "Extract facts as JSON: "
|
||||
|
||||
messages = [{"role": "user", "content": "Hello there"}]
|
||||
|
||||
facts = await store.extract_facts(messages, provider, "claude-sonnet-4-6")
|
||||
|
||||
assert facts == []
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_extract_facts_skips_empty_messages():
|
||||
"""extract_facts should return empty list when all messages have empty content."""
|
||||
try:
|
||||
from nanobot.agent.memory_mem0 import Mem0MemoryStore
|
||||
except ImportError:
|
||||
pytest.skip("mem0 not installed")
|
||||
|
||||
provider = AsyncMock()
|
||||
|
||||
store = object.__new__(Mem0MemoryStore)
|
||||
store.custom_prompt = "Extract facts as JSON: "
|
||||
|
||||
messages = [
|
||||
{"role": "user", "content": ""},
|
||||
{"role": "assistant", "content": ""},
|
||||
]
|
||||
|
||||
facts = await store.extract_facts(messages, provider, "claude-sonnet-4-6")
|
||||
|
||||
assert facts == []
|
||||
# Provider should not be called when there's no content
|
||||
provider.chat.assert_not_called()
|
||||
@@ -0,0 +1,70 @@
|
||||
"""Security tests for MemoryTool20250818."""
|
||||
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
from nanobot.agent.tools.anthropic import MemoryTool20250818
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def temp_workspace(tmp_path):
|
||||
"""Create temporary workspace."""
|
||||
return tmp_path
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def memory_tool(temp_workspace):
|
||||
"""Create MemoryTool instance."""
|
||||
return MemoryTool20250818(workspace=temp_workspace)
|
||||
|
||||
|
||||
class TestPathSecurity:
|
||||
"""Test path validation security."""
|
||||
|
||||
def test_validate_path_valid_root(self, memory_tool):
|
||||
"""Test that /memories is valid."""
|
||||
result = memory_tool._validate_memory_path("/memories")
|
||||
assert result == memory_tool.memories_dir
|
||||
|
||||
def test_validate_path_valid_file(self, memory_tool):
|
||||
"""Test that /memories/notes.txt is valid."""
|
||||
result = memory_tool._validate_memory_path("/memories/notes.txt")
|
||||
assert result == memory_tool.memories_dir / "notes.txt"
|
||||
|
||||
def test_validate_path_valid_nested(self, memory_tool):
|
||||
"""Test that /memories/project/status.xml is valid."""
|
||||
result = memory_tool._validate_memory_path("/memories/project/status.xml")
|
||||
assert result == memory_tool.memories_dir / "project" / "status.xml"
|
||||
|
||||
def test_validate_path_rejects_parent_traversal(self, memory_tool):
|
||||
"""Test that ../ is rejected."""
|
||||
with pytest.raises(ValueError, match="escapes /memories directory"):
|
||||
memory_tool._validate_memory_path("/memories/../config.json")
|
||||
|
||||
def test_validate_path_rejects_double_parent_traversal(self, memory_tool):
|
||||
"""Test that ../../ is rejected."""
|
||||
with pytest.raises(ValueError, match="escapes /memories directory"):
|
||||
memory_tool._validate_memory_path("/memories/../../etc/passwd")
|
||||
|
||||
def test_validate_path_rejects_absolute_path(self, memory_tool):
|
||||
"""Test that absolute paths are rejected."""
|
||||
with pytest.raises(ValueError, match="must start with /memories"):
|
||||
memory_tool._validate_memory_path("/etc/passwd")
|
||||
|
||||
def test_validate_path_rejects_workspace_path(self, memory_tool):
|
||||
"""Test that /workspace paths are rejected."""
|
||||
with pytest.raises(ValueError, match="must start with /memories"):
|
||||
memory_tool._validate_memory_path("/workspace/data.txt")
|
||||
|
||||
def test_validate_path_rejects_relative_path(self, memory_tool):
|
||||
"""Test that relative paths are rejected."""
|
||||
with pytest.raises(ValueError, match="must start with /memories"):
|
||||
memory_tool._validate_memory_path("notes.txt")
|
||||
|
||||
def test_validate_path_url_encoded_is_safe(self, memory_tool):
|
||||
"""Test that URL-encoded paths are safe (not decoded by pathlib)."""
|
||||
# Python's pathlib treats %2e%2e as literal characters, not as ..
|
||||
# So this is actually safe - it creates a subdirectory named "%2e%2e"
|
||||
attack_path = "/memories/%2e%2e/config.json"
|
||||
result = memory_tool._validate_memory_path(attack_path)
|
||||
# This should resolve to memories/%2e%2e/config.json (literal characters)
|
||||
assert result == memory_tool.memories_dir / "%2e%2e" / "config.json"
|
||||
@@ -0,0 +1,239 @@
|
||||
"""Tests for MemoryTool20250818."""
|
||||
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
from nanobot.agent.tools.anthropic import MemoryTool20250818
|
||||
from nanobot.agent.tools.anthropic.base import CLIResult
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def temp_workspace(tmp_path):
|
||||
"""Create temporary workspace."""
|
||||
return tmp_path
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def memory_tool(temp_workspace):
|
||||
"""Create MemoryTool instance."""
|
||||
return MemoryTool20250818(workspace=temp_workspace)
|
||||
|
||||
|
||||
def test_memory_tool_initialization(memory_tool, temp_workspace):
|
||||
"""Test that MemoryTool initializes correctly."""
|
||||
assert memory_tool.api_type == "memory_20250818"
|
||||
assert memory_tool.name == "memory"
|
||||
assert memory_tool.beta_flag == "context-management-2025-06-27"
|
||||
assert (temp_workspace / "memories").exists()
|
||||
|
||||
|
||||
def test_memory_tool_to_params(memory_tool):
|
||||
"""Test that to_params returns correct format."""
|
||||
params = memory_tool.to_params()
|
||||
assert params == {
|
||||
"type": "memory_20250818",
|
||||
"name": "memory"
|
||||
}
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_view_file(memory_tool, temp_workspace):
|
||||
"""Test viewing a file with line numbers."""
|
||||
# Create test file
|
||||
test_file = temp_workspace / "memories" / "notes.txt"
|
||||
test_file.write_text("Line 1\nLine 2\nLine 3\n")
|
||||
|
||||
result = await memory_tool(command="view", path="/memories/notes.txt")
|
||||
|
||||
assert result.exit_code == 0
|
||||
assert result.error == ""
|
||||
assert "Here's the content of /memories/notes.txt with line numbers:" in result.output
|
||||
assert " 1\tLine 1" in result.output
|
||||
assert " 2\tLine 2" in result.output
|
||||
assert " 3\tLine 3" in result.output
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_view_file_with_range(memory_tool, temp_workspace):
|
||||
"""Test viewing a file with line range."""
|
||||
# Create test file with 10 lines
|
||||
test_file = temp_workspace / "memories" / "test.txt"
|
||||
test_file.write_text("\n".join([f"Line {i}" for i in range(1, 11)]))
|
||||
|
||||
result = await memory_tool(command="view", path="/memories/test.txt", view_range=[3, 5])
|
||||
|
||||
assert result.exit_code == 0
|
||||
assert " 3\tLine 3" in result.output
|
||||
assert " 4\tLine 4" in result.output
|
||||
assert " 5\tLine 5" in result.output
|
||||
assert "Line 1" not in result.output
|
||||
assert "Line 10" not in result.output
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_view_file_not_exists(memory_tool):
|
||||
"""Test viewing a nonexistent file."""
|
||||
result = await memory_tool(command="view", path="/memories/nonexistent.txt")
|
||||
|
||||
assert result.exit_code == 1
|
||||
assert result.output == ""
|
||||
assert "The path /memories/nonexistent.txt does not exist" in result.error
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_view_directory(memory_tool, temp_workspace):
|
||||
"""Test viewing a directory listing."""
|
||||
# Create test directory structure
|
||||
memories = temp_workspace / "memories"
|
||||
(memories / "notes.txt").write_text("content")
|
||||
(memories / "project").mkdir()
|
||||
(memories / "project" / "status.xml").write_text("<status>ok</status>")
|
||||
(memories / ".hidden").write_text("hidden") # Should be excluded
|
||||
|
||||
result = await memory_tool(command="view", path="/memories")
|
||||
|
||||
assert result.exit_code == 0
|
||||
assert result.error == ""
|
||||
assert "Here're the files and directories up to 2 levels deep in /memories" in result.output
|
||||
assert "/memories" in result.output
|
||||
assert "/memories/notes.txt" in result.output
|
||||
assert "/memories/project" in result.output
|
||||
assert "/memories/project/status.xml" in result.output
|
||||
assert ".hidden" not in result.output # Hidden files excluded
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_view_empty_directory(memory_tool):
|
||||
"""Test viewing an empty directory."""
|
||||
result = await memory_tool(command="view", path="/memories")
|
||||
|
||||
assert result.exit_code == 0
|
||||
assert "/memories" in result.output
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_create_file(memory_tool, temp_workspace):
|
||||
"""Test creating a new file."""
|
||||
result = await memory_tool(
|
||||
command="create",
|
||||
path="/memories/notes.txt",
|
||||
file_text="My notes\nLine 2\n"
|
||||
)
|
||||
|
||||
assert result.exit_code == 0
|
||||
assert result.error == ""
|
||||
assert "File created successfully at: /memories/notes.txt" in result.output
|
||||
|
||||
# Verify file was created
|
||||
created_file = temp_workspace / "memories" / "notes.txt"
|
||||
assert created_file.exists()
|
||||
assert created_file.read_text() == "My notes\nLine 2\n"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_create_file_nested_directory(memory_tool, temp_workspace):
|
||||
"""Test creating a file in a nested directory (auto-creates parent dirs)."""
|
||||
result = await memory_tool(
|
||||
command="create",
|
||||
path="/memories/project/status.xml",
|
||||
file_text="<status>ok</status>"
|
||||
)
|
||||
|
||||
assert result.exit_code == 0
|
||||
assert "File created successfully at: /memories/project/status.xml" in result.output
|
||||
|
||||
# Verify file and parent directory were created
|
||||
created_file = temp_workspace / "memories" / "project" / "status.xml"
|
||||
assert created_file.exists()
|
||||
assert created_file.read_text() == "<status>ok</status>"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_create_file_already_exists(memory_tool, temp_workspace):
|
||||
"""Test creating a file that already exists."""
|
||||
# Create file first
|
||||
existing = temp_workspace / "memories" / "existing.txt"
|
||||
existing.write_text("existing content")
|
||||
|
||||
result = await memory_tool(
|
||||
command="create",
|
||||
path="/memories/existing.txt",
|
||||
file_text="new content"
|
||||
)
|
||||
|
||||
assert result.exit_code == 1
|
||||
assert result.output == ""
|
||||
assert "Error: File /memories/existing.txt already exists" in result.error
|
||||
|
||||
# Verify original content unchanged
|
||||
assert existing.read_text() == "existing content"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_create_file_missing_text(memory_tool):
|
||||
"""Test creating a file without file_text parameter."""
|
||||
result = await memory_tool(
|
||||
command="create",
|
||||
path="/memories/notes.txt"
|
||||
)
|
||||
|
||||
assert result.exit_code == 1
|
||||
assert result.output == ""
|
||||
assert "Error: file_text is required for create command" in result.error
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_str_replace_success(memory_tool, temp_workspace):
|
||||
"""Test replacing unique string in a file."""
|
||||
test_file = temp_workspace / "memories" / "config.txt"
|
||||
test_file.write_text("color: blue\nsize: large\n")
|
||||
|
||||
result = await memory_tool(
|
||||
command="str_replace",
|
||||
path="/memories/config.txt",
|
||||
old_str="blue",
|
||||
new_str="green"
|
||||
)
|
||||
|
||||
assert result.exit_code == 0
|
||||
assert result.error == ""
|
||||
assert "The memory file has been edited." in result.output
|
||||
|
||||
# Verify file was modified
|
||||
assert test_file.read_text() == "color: green\nsize: large\n"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_str_replace_not_found(memory_tool, temp_workspace):
|
||||
"""Test replacing string that doesn't exist."""
|
||||
test_file = temp_workspace / "memories" / "config.txt"
|
||||
test_file.write_text("color: blue\n")
|
||||
|
||||
result = await memory_tool(
|
||||
command="str_replace",
|
||||
path="/memories/config.txt",
|
||||
old_str="red",
|
||||
new_str="green"
|
||||
)
|
||||
|
||||
assert result.exit_code == 1
|
||||
assert result.output == ""
|
||||
assert "No replacement was performed, old_str `red` did not appear verbatim" in result.error
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_str_replace_duplicate(memory_tool, temp_workspace):
|
||||
"""Test replacing string that appears multiple times."""
|
||||
test_file = temp_workspace / "memories" / "config.txt"
|
||||
test_file.write_text("color: blue\nbackground: blue\n")
|
||||
|
||||
result = await memory_tool(
|
||||
command="str_replace",
|
||||
path="/memories/config.txt",
|
||||
old_str="blue",
|
||||
new_str="green"
|
||||
)
|
||||
|
||||
assert result.exit_code == 1
|
||||
assert result.output == ""
|
||||
assert "Multiple occurrences of old_str `blue`" in result.error
|
||||
assert "Please ensure it is unique" in result.error
|
||||
@@ -0,0 +1,153 @@
|
||||
"""Tests for message visibility signing (hidden intermediate messages)."""
|
||||
|
||||
import json
|
||||
from pathlib import Path
|
||||
|
||||
from nanobot.agent.context import ContextBuilder
|
||||
from nanobot.agent.visibility import compute_signature, sign_content
|
||||
from nanobot.session.manager import Session
|
||||
|
||||
|
||||
class TestComputeSignature:
|
||||
"""Tests for compute_signature()."""
|
||||
|
||||
def test_returns_8_char_hex(self):
|
||||
sig = compute_signature("hello")
|
||||
assert len(sig) == 8
|
||||
assert all(c in "0123456789abcdef" for c in sig)
|
||||
|
||||
def test_deterministic(self):
|
||||
assert compute_signature("hello") == compute_signature("hello")
|
||||
|
||||
def test_different_content_different_sig(self):
|
||||
assert compute_signature("hello") != compute_signature("world")
|
||||
|
||||
def test_sign_content_uses_compute_signature(self):
|
||||
"""sign_content should produce [HIDDEN:{compute_signature(content)}] prefix."""
|
||||
content = "test message"
|
||||
sig = compute_signature(content)
|
||||
assert sign_content(content) == f"[HIDDEN:{sig}] {content}"
|
||||
|
||||
|
||||
class TestAddAssistantMessage:
|
||||
"""Tests for _hidden_sig in add_assistant_message()."""
|
||||
|
||||
def setup_method(self):
|
||||
self.ctx = ContextBuilder(Path("/tmp"))
|
||||
|
||||
def test_intermediate_message_gets_hidden_sig(self):
|
||||
msgs: list = []
|
||||
tool_calls = [{"id": "tc1", "type": "function", "function": {"name": "test", "arguments": "{}"}}]
|
||||
self.ctx.add_assistant_message(msgs, "thinking...", tool_calls)
|
||||
|
||||
assert msgs[0].get("_hidden_sig") is not None
|
||||
assert msgs[0]["_hidden_sig"] == compute_signature("thinking...")
|
||||
|
||||
def test_final_message_no_hidden_sig(self):
|
||||
msgs: list = []
|
||||
self.ctx.add_assistant_message(msgs, "Here is the answer", None)
|
||||
|
||||
assert "_hidden_sig" not in msgs[0]
|
||||
|
||||
def test_empty_content_signed(self):
|
||||
msgs: list = []
|
||||
tool_calls = [{"id": "tc1", "type": "function", "function": {"name": "test", "arguments": "{}"}}]
|
||||
self.ctx.add_assistant_message(msgs, None, tool_calls)
|
||||
|
||||
assert msgs[0]["_hidden_sig"] == compute_signature("")
|
||||
|
||||
|
||||
class TestAddToolResult:
|
||||
"""Tests for _hidden_sig in add_tool_result()."""
|
||||
|
||||
def setup_method(self):
|
||||
self.ctx = ContextBuilder(Path("/tmp"))
|
||||
|
||||
def test_tool_result_gets_hidden_sig(self):
|
||||
msgs: list = []
|
||||
self.ctx.add_tool_result(msgs, "tc1", "read_file", "file contents here")
|
||||
|
||||
assert msgs[0]["_hidden_sig"] == compute_signature("file contents here")
|
||||
|
||||
def test_tool_result_non_string_content(self):
|
||||
msgs: list = []
|
||||
# Multipart content (e.g. image) is a list, not a string
|
||||
self.ctx.add_tool_result(msgs, "tc1", "screenshot", [{"type": "text", "text": "ok"}])
|
||||
|
||||
assert msgs[0]["_hidden_sig"] == compute_signature("")
|
||||
|
||||
|
||||
class TestGetHistoryPrefix:
|
||||
"""Tests for get_history() applying [HIDDEN:sig] prefix."""
|
||||
|
||||
def test_hidden_sig_applied_at_read_time(self):
|
||||
session = Session(key="test")
|
||||
sig = compute_signature("thinking...")
|
||||
session.messages = [
|
||||
{"role": "assistant", "content": "thinking...", "tool_calls": [{}], "_hidden_sig": sig},
|
||||
]
|
||||
|
||||
history = session.get_history()
|
||||
assert history[0]["content"] == f"[HIDDEN:{sig}] thinking..."
|
||||
assert "_hidden_sig" not in history[0]
|
||||
|
||||
def test_no_prefix_without_hidden_sig(self):
|
||||
session = Session(key="test")
|
||||
session.messages = [
|
||||
{"role": "assistant", "content": "Here is the answer"},
|
||||
]
|
||||
|
||||
history = session.get_history()
|
||||
assert history[0]["content"] == "Here is the answer"
|
||||
|
||||
def test_tool_result_gets_prefix(self):
|
||||
session = Session(key="test")
|
||||
sig = compute_signature("file contents")
|
||||
session.messages = [
|
||||
{"role": "tool", "tool_call_id": "tc1", "name": "read", "content": "file contents", "_hidden_sig": sig},
|
||||
]
|
||||
|
||||
history = session.get_history()
|
||||
assert history[0]["content"] == f"[HIDDEN:{sig}] file contents"
|
||||
|
||||
def test_roundtrip_jsonl(self, tmp_path):
|
||||
"""Write to session JSONL, reload, verify get_history() produces correct prefix."""
|
||||
from nanobot.session.manager import SessionManager
|
||||
|
||||
workspace = tmp_path / "workspace"
|
||||
workspace.mkdir()
|
||||
mgr = SessionManager(workspace)
|
||||
|
||||
session = mgr.get_or_create("test:roundtrip")
|
||||
sig = compute_signature("intermediate")
|
||||
session.add_raw_message({
|
||||
"role": "assistant",
|
||||
"content": "intermediate",
|
||||
"tool_calls": [{"id": "tc1", "type": "function", "function": {"name": "x", "arguments": "{}"}}],
|
||||
"_hidden_sig": sig,
|
||||
})
|
||||
session.add_raw_message({
|
||||
"role": "assistant",
|
||||
"content": "final answer",
|
||||
})
|
||||
mgr.save(session)
|
||||
|
||||
# Reload from disk
|
||||
mgr.invalidate("test:roundtrip")
|
||||
reloaded = mgr.get_or_create("test:roundtrip")
|
||||
history = reloaded.get_history()
|
||||
|
||||
assert history[0]["content"] == f"[HIDDEN:{sig}] intermediate"
|
||||
assert history[1]["content"] == "final answer"
|
||||
|
||||
def test_idempotent_across_calls(self):
|
||||
"""Same prefix produced every call (cache stability)."""
|
||||
session = Session(key="test")
|
||||
sig = compute_signature("msg")
|
||||
session.messages = [
|
||||
{"role": "assistant", "content": "msg", "_hidden_sig": sig},
|
||||
]
|
||||
|
||||
h1 = session.get_history()
|
||||
h2 = session.get_history()
|
||||
assert h1[0]["content"] == h2[0]["content"]
|
||||
@@ -0,0 +1,54 @@
|
||||
"""Test registration of native Anthropic tools in the agent loop."""
|
||||
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.agent.tools.anthropic import (
|
||||
BashTool20250124,
|
||||
EditTool20250728,
|
||||
ComputerTool20251124,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_provider():
|
||||
"""Create a mock provider."""
|
||||
provider = MagicMock()
|
||||
provider.chat = AsyncMock(return_value="test response")
|
||||
provider.get_default_model = MagicMock(return_value="test-model")
|
||||
return provider
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_bus():
|
||||
"""Create a mock message bus."""
|
||||
bus = MagicMock()
|
||||
bus.publish_outbound = AsyncMock()
|
||||
return bus
|
||||
|
||||
|
||||
def test_native_tools_registered(mock_provider, mock_bus, tmp_path):
|
||||
"""Test that native Anthropic tools are registered in the agent loop."""
|
||||
# Create agent loop
|
||||
loop = AgentLoop(
|
||||
provider=mock_provider,
|
||||
bus=mock_bus,
|
||||
workspace=tmp_path,
|
||||
)
|
||||
|
||||
# Get all registered tool names
|
||||
tool_names = [tool.name for tool in loop.tools._tools.values()]
|
||||
|
||||
# Verify native tools are registered (using their internal names)
|
||||
assert "bash" in tool_names, "bash tool should be registered"
|
||||
assert "str_replace_based_edit_tool" in tool_names, "str_replace_based_edit_tool tool should be registered"
|
||||
# Note: computer tool is intentionally disabled by default (requires VNC setup)
|
||||
|
||||
# Verify we can get the tool instances
|
||||
bash_tool = loop.tools.get("bash")
|
||||
assert isinstance(bash_tool, BashTool20250124)
|
||||
|
||||
editor_tool = loop.tools.get("str_replace_based_edit_tool")
|
||||
assert isinstance(editor_tool, EditTool20250728)
|
||||
@@ -0,0 +1,48 @@
|
||||
"""Test OAuth configuration schema."""
|
||||
import pytest
|
||||
from nanobot.config.schema import ProviderConfig, OAuthCredentials
|
||||
|
||||
|
||||
def test_provider_config_has_oauth_fields():
|
||||
"""ProviderConfig should have oauth_credentials field."""
|
||||
config = ProviderConfig(api_key="test")
|
||||
assert hasattr(config, "oauth_credentials")
|
||||
assert config.oauth_credentials is None
|
||||
|
||||
|
||||
def test_oauth_credentials_model():
|
||||
"""OAuthCredentials should store token, refresh, expiry."""
|
||||
creds = OAuthCredentials(
|
||||
access_token="sk-ant-oat01-xxx",
|
||||
refresh_token="rt_xxx",
|
||||
expires_at=1234567890,
|
||||
token_type="oauth"
|
||||
)
|
||||
assert creds.access_token.startswith("sk-ant-oat")
|
||||
assert creds.is_oauth_token is True
|
||||
|
||||
|
||||
def test_oauth_credentials_expiry_check():
|
||||
"""OAuthCredentials should detect expired tokens."""
|
||||
import time
|
||||
expired = OAuthCredentials(
|
||||
access_token="sk-ant-oat01-xxx",
|
||||
expires_at=int(time.time()) - 3600 # 1 hour ago
|
||||
)
|
||||
assert expired.is_expired is True
|
||||
|
||||
valid = OAuthCredentials(
|
||||
access_token="sk-ant-oat01-xxx",
|
||||
expires_at=int(time.time()) + 3600 # 1 hour from now
|
||||
)
|
||||
assert valid.is_expired is False
|
||||
|
||||
|
||||
def test_oauth_credentials_no_expiry():
|
||||
"""Setup tokens with expires_at=0 should never be expired."""
|
||||
creds = OAuthCredentials(
|
||||
access_token="sk-ant-oat01-xxx",
|
||||
expires_at=0 # No expiry (setup-token)
|
||||
)
|
||||
assert creds.is_expired is False
|
||||
assert creds.expires_soon is False
|
||||
@@ -0,0 +1,102 @@
|
||||
"""Test that the Anthropic OAuth identity block is always included in API requests."""
|
||||
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, patch, MagicMock
|
||||
|
||||
import httpx
|
||||
|
||||
from nanobot.providers.anthropic_oauth import AnthropicOAuthProvider
|
||||
from nanobot.providers.oauth_utils import get_claude_code_system_prefix
|
||||
|
||||
|
||||
IDENTITY_TEXT = get_claude_code_system_prefix()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def provider():
|
||||
return AnthropicOAuthProvider(
|
||||
oauth_token="sk-ant-oat01-test-token",
|
||||
default_model="claude-opus-4-7",
|
||||
)
|
||||
|
||||
|
||||
def _mock_response(status_code=200, json_data=None):
|
||||
"""Create a mock httpx.Response."""
|
||||
resp = MagicMock(spec=httpx.Response)
|
||||
resp.status_code = status_code
|
||||
resp.headers = {}
|
||||
resp.text = ""
|
||||
resp.json.return_value = json_data or {
|
||||
"content": [{"type": "text", "text": "ok"}],
|
||||
"stop_reason": "end_turn",
|
||||
"usage": {"input_tokens": 1, "output_tokens": 1},
|
||||
}
|
||||
return resp
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_identity_block_present_with_system_prompt(provider):
|
||||
"""When a system prompt is provided, identity block is the first system block."""
|
||||
mock_client = AsyncMock()
|
||||
mock_client.post.return_value = _mock_response()
|
||||
provider._client = mock_client
|
||||
|
||||
await provider._make_request(
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
system="You are a helpful assistant.",
|
||||
)
|
||||
|
||||
call_kwargs = mock_client.post.call_args
|
||||
payload = call_kwargs.kwargs["json"] if "json" in call_kwargs.kwargs else call_kwargs[1]["json"]
|
||||
system_blocks = payload["system"]
|
||||
|
||||
assert len(system_blocks) == 2
|
||||
assert system_blocks[0]["type"] == "text"
|
||||
assert system_blocks[0]["text"] == IDENTITY_TEXT
|
||||
assert system_blocks[1]["text"] == "You are a helpful assistant."
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_identity_block_present_without_system_prompt(provider):
|
||||
"""When no system prompt is provided, identity block is still included.
|
||||
|
||||
This is the critical fix: extract_facts and similar calls pass system=None,
|
||||
but Anthropic requires the identity block for OAuth tokens.
|
||||
"""
|
||||
mock_client = AsyncMock()
|
||||
mock_client.post.return_value = _mock_response()
|
||||
provider._client = mock_client
|
||||
|
||||
await provider._make_request(
|
||||
messages=[{"role": "user", "content": "extract facts"}],
|
||||
system=None,
|
||||
)
|
||||
|
||||
call_kwargs = mock_client.post.call_args
|
||||
payload = call_kwargs.kwargs["json"] if "json" in call_kwargs.kwargs else call_kwargs[1]["json"]
|
||||
system_blocks = payload["system"]
|
||||
|
||||
assert len(system_blocks) == 1
|
||||
assert system_blocks[0]["type"] == "text"
|
||||
assert system_blocks[0]["text"] == IDENTITY_TEXT
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_identity_block_present_with_empty_string_system(provider):
|
||||
"""Empty string system prompt should still include the identity block."""
|
||||
mock_client = AsyncMock()
|
||||
mock_client.post.return_value = _mock_response()
|
||||
provider._client = mock_client
|
||||
|
||||
await provider._make_request(
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
system="",
|
||||
)
|
||||
|
||||
call_kwargs = mock_client.post.call_args
|
||||
payload = call_kwargs.kwargs["json"] if "json" in call_kwargs.kwargs else call_kwargs[1]["json"]
|
||||
system_blocks = payload["system"]
|
||||
|
||||
# Empty string is falsy, so should go through the else branch
|
||||
assert len(system_blocks) == 1
|
||||
assert system_blocks[0]["text"] == IDENTITY_TEXT
|
||||
@@ -0,0 +1,55 @@
|
||||
"""Test OAuth credential storage."""
|
||||
import pytest
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from nanobot.config.oauth_store import OAuthStore
|
||||
from nanobot.config.schema import OAuthCredentials
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def temp_store():
|
||||
"""Create store with temp directory."""
|
||||
with tempfile.TemporaryDirectory() as tmpdir:
|
||||
yield OAuthStore(Path(tmpdir) / ".nanobot")
|
||||
|
||||
|
||||
def test_save_and_load_credentials(temp_store):
|
||||
"""Should save and load OAuth credentials."""
|
||||
creds = OAuthCredentials(
|
||||
access_token="sk-ant-oat01-xxx",
|
||||
refresh_token="rt_xxx",
|
||||
expires_at=1234567890
|
||||
)
|
||||
|
||||
temp_store.save("anthropic", creds)
|
||||
loaded = temp_store.load("anthropic")
|
||||
|
||||
assert loaded is not None
|
||||
assert loaded.access_token == creds.access_token
|
||||
assert loaded.refresh_token == creds.refresh_token
|
||||
|
||||
|
||||
def test_load_nonexistent_returns_none(temp_store):
|
||||
"""Should return None for missing credentials."""
|
||||
assert temp_store.load("nonexistent") is None
|
||||
|
||||
|
||||
def test_delete_credentials(temp_store):
|
||||
"""Should delete saved credentials."""
|
||||
creds = OAuthCredentials(access_token="sk-ant-oat01-xxx")
|
||||
temp_store.save("anthropic", creds)
|
||||
assert temp_store.delete("anthropic") is True
|
||||
assert temp_store.load("anthropic") is None
|
||||
|
||||
|
||||
def test_delete_nonexistent_returns_false(temp_store):
|
||||
"""Should return False when deleting missing credentials."""
|
||||
assert temp_store.delete("nonexistent") is False
|
||||
|
||||
|
||||
def test_file_permissions(temp_store):
|
||||
"""Credentials file should have restricted permissions."""
|
||||
creds = OAuthCredentials(access_token="sk-ant-oat01-xxx")
|
||||
temp_store.save("anthropic", creds)
|
||||
perms = oct(temp_store.file_path.stat().st_mode)[-3:]
|
||||
assert perms == "600"
|
||||
@@ -0,0 +1,28 @@
|
||||
"""Test OAuth utility functions."""
|
||||
import pytest
|
||||
from nanobot.providers.oauth_utils import is_oauth_token, get_auth_headers
|
||||
|
||||
|
||||
def test_is_oauth_token_detects_oat():
|
||||
"""Should detect sk-ant-oat tokens as OAuth."""
|
||||
assert is_oauth_token("sk-ant-oat01-buSdhCH2XEkebW7ZQZTvGqH5EwAFh4u52LrdJhAP") is True
|
||||
assert is_oauth_token("sk-ant-api03-regularkey") is False
|
||||
assert is_oauth_token("") is False
|
||||
assert is_oauth_token(None) is False
|
||||
|
||||
|
||||
def test_get_auth_headers_oauth():
|
||||
"""OAuth tokens should use Authorization: Bearer."""
|
||||
headers = get_auth_headers("sk-ant-oat01-xxx", is_oauth=True)
|
||||
assert "Authorization" in headers
|
||||
assert headers["Authorization"] == "Bearer sk-ant-oat01-xxx"
|
||||
assert "x-api-key" not in headers
|
||||
assert headers["anthropic-beta"] == "claude-code-20250219,oauth-2025-04-20,context-management-2025-06-27"
|
||||
|
||||
|
||||
def test_get_auth_headers_api_key():
|
||||
"""Regular API keys should use x-api-key."""
|
||||
headers = get_auth_headers("sk-ant-api03-xxx", is_oauth=False)
|
||||
assert "x-api-key" in headers
|
||||
assert headers["x-api-key"] == "sk-ant-api03-xxx"
|
||||
assert "Authorization" not in headers
|
||||
@@ -0,0 +1,24 @@
|
||||
"""Tests for correlation resolution in outbound dispatch."""
|
||||
|
||||
import asyncio
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.bus.events import OutboundMessage
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_dispatch_resolves_correlation_before_channel_send():
|
||||
"""Correlation Future should be resolved when outbound message is dispatched."""
|
||||
bus = MessageBus()
|
||||
future = bus.register_correlation("corr-1")
|
||||
|
||||
msg = OutboundMessage(channel="telegram", chat_id="123", content="response", metadata={"correlation_id": "corr-1"})
|
||||
await bus.publish_outbound(msg)
|
||||
|
||||
# Simulate what _dispatch_outbound does: consume + resolve
|
||||
consumed = await asyncio.wait_for(bus.consume_outbound(), timeout=1.0)
|
||||
bus.resolve_correlation(consumed)
|
||||
|
||||
assert future.done()
|
||||
assert future.result() == "response"
|
||||
@@ -0,0 +1,32 @@
|
||||
"""Test provider factory with OAuth support."""
|
||||
import pytest
|
||||
from nanobot.providers import create_provider
|
||||
from nanobot.providers.anthropic_oauth import AnthropicOAuthProvider
|
||||
from nanobot.providers.litellm_provider import LiteLLMProvider
|
||||
|
||||
|
||||
def test_create_provider_oauth_token():
|
||||
"""OAuth tokens should create AnthropicOAuthProvider."""
|
||||
provider = create_provider(
|
||||
api_key="sk-ant-oat01-test-token",
|
||||
model="anthropic/claude-opus-4-7"
|
||||
)
|
||||
assert isinstance(provider, AnthropicOAuthProvider)
|
||||
|
||||
|
||||
def test_create_provider_regular_key():
|
||||
"""Regular API keys should create LiteLLMProvider."""
|
||||
provider = create_provider(
|
||||
api_key="sk-ant-api03-regular-key",
|
||||
model="anthropic/claude-opus-4-7"
|
||||
)
|
||||
assert isinstance(provider, LiteLLMProvider)
|
||||
|
||||
|
||||
def test_create_provider_openrouter():
|
||||
"""OpenRouter keys should create LiteLLMProvider."""
|
||||
provider = create_provider(
|
||||
api_key="sk-or-v1-xxx",
|
||||
model="anthropic/claude-opus-4-7"
|
||||
)
|
||||
assert isinstance(provider, LiteLLMProvider)
|
||||
@@ -0,0 +1,94 @@
|
||||
"""Tests for registry duck typing support."""
|
||||
|
||||
import pytest
|
||||
from nanobot.agent.tools.registry import ToolRegistry
|
||||
from nanobot.agent.tools.anthropic.base import BaseAnthropicTool, ToolResult
|
||||
|
||||
|
||||
class MockNativeTool(BaseAnthropicTool):
|
||||
"""Mock native tool for testing."""
|
||||
api_type = "test_20250227"
|
||||
name = "native_test"
|
||||
beta_flag = "test-beta"
|
||||
|
||||
async def __call__(self, **kwargs):
|
||||
return ToolResult(output="native result")
|
||||
|
||||
def to_params(self):
|
||||
return {"type": self.api_type, "name": self.name}
|
||||
|
||||
|
||||
class MockFunctionTool:
|
||||
"""Mock function tool for testing."""
|
||||
def __init__(self):
|
||||
self.name = "function_test"
|
||||
|
||||
def to_schema(self):
|
||||
return {
|
||||
"type": "function",
|
||||
"function": {
|
||||
"name": self.name,
|
||||
"description": "Test function tool",
|
||||
"parameters": {"type": "object", "properties": {}}
|
||||
}
|
||||
}
|
||||
|
||||
async def execute(self, **kwargs):
|
||||
return "function result"
|
||||
|
||||
|
||||
def test_registry_supports_native_tools():
|
||||
"""Test registry can register and get definitions from native tools."""
|
||||
registry = ToolRegistry()
|
||||
native_tool = MockNativeTool()
|
||||
registry.register(native_tool)
|
||||
|
||||
definitions = registry.get_definitions()
|
||||
assert len(definitions) == 1
|
||||
assert definitions[0]["type"] == "test_20250227"
|
||||
assert definitions[0]["name"] == "native_test"
|
||||
|
||||
|
||||
def test_registry_supports_function_tools():
|
||||
"""Test registry still supports function tools."""
|
||||
registry = ToolRegistry()
|
||||
function_tool = MockFunctionTool()
|
||||
registry.register(function_tool)
|
||||
|
||||
definitions = registry.get_definitions()
|
||||
assert len(definitions) == 1
|
||||
assert definitions[0]["type"] == "function"
|
||||
assert definitions[0]["function"]["name"] == "function_test"
|
||||
|
||||
|
||||
def test_registry_supports_mixed_tools():
|
||||
"""Test registry can handle both native and function tools."""
|
||||
registry = ToolRegistry()
|
||||
native_tool = MockNativeTool()
|
||||
function_tool = MockFunctionTool()
|
||||
|
||||
registry.register(native_tool)
|
||||
registry.register(function_tool)
|
||||
|
||||
definitions = registry.get_definitions()
|
||||
assert len(definitions) == 2
|
||||
|
||||
# Find each tool type in definitions
|
||||
native_def = next(d for d in definitions if d.get("type") == "test_20250227")
|
||||
function_def = next(d for d in definitions if d.get("type") == "function")
|
||||
|
||||
assert native_def["name"] == "native_test"
|
||||
assert function_def["function"]["name"] == "function_test"
|
||||
|
||||
|
||||
def test_registry_rejects_tools_without_schema_method():
|
||||
"""Test registry raises error for tools with no schema method."""
|
||||
registry = ToolRegistry()
|
||||
|
||||
class BadTool:
|
||||
name = "bad"
|
||||
|
||||
registry.register(BadTool())
|
||||
|
||||
with pytest.raises(ValueError, match="has no schema method"):
|
||||
registry.get_definitions()
|
||||
@@ -0,0 +1,57 @@
|
||||
"""Tests for registry execution of native tools."""
|
||||
|
||||
import pytest
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
from nanobot.agent.tools.registry import ToolRegistry
|
||||
from nanobot.agent.tools.anthropic import BashTool20250124, EditTool20250728
|
||||
from nanobot.agent.tools.anthropic.base import ToolResult, CLIResult
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_registry_executes_bash_tool():
|
||||
"""Test registry can execute BashTool20250124 and returns ToolResult."""
|
||||
registry = ToolRegistry()
|
||||
registry.register(BashTool20250124())
|
||||
|
||||
result = await registry.execute("bash", {"command": "echo 'test'"})
|
||||
|
||||
assert isinstance(result, ToolResult)
|
||||
assert result.output is not None
|
||||
assert "test" in result.output
|
||||
assert result.error is None
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_registry_executes_edit_tool():
|
||||
"""Test registry can execute EditTool20250728 and returns CLIResult."""
|
||||
registry = ToolRegistry()
|
||||
registry.register(EditTool20250728())
|
||||
|
||||
with tempfile.TemporaryDirectory() as tmpdir:
|
||||
test_file = str(Path(tmpdir) / "test.txt")
|
||||
|
||||
result = await registry.execute("str_replace_based_edit_tool", {
|
||||
"command": "create",
|
||||
"path": test_file,
|
||||
"file_text": "Hello, world!"
|
||||
})
|
||||
|
||||
assert isinstance(result, CLIResult)
|
||||
assert "created" in result.output.lower() or "success" in result.output.lower()
|
||||
assert Path(test_file).exists()
|
||||
assert Path(test_file).read_text() == "Hello, world!"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_registry_mixed_tools():
|
||||
"""Test registry can execute both native and function tools in same registry."""
|
||||
registry = ToolRegistry()
|
||||
|
||||
# Register native tool
|
||||
registry.register(BashTool20250124())
|
||||
|
||||
# Execute native tool
|
||||
result = await registry.execute("bash", {"command": "echo 'native'"})
|
||||
assert isinstance(result, ToolResult)
|
||||
assert "native" in result.output
|
||||
@@ -0,0 +1,21 @@
|
||||
"""Test OAuth detection in provider registry."""
|
||||
import pytest
|
||||
from nanobot.providers.registry import should_use_oauth_provider
|
||||
|
||||
|
||||
def test_should_use_oauth_for_oat_token():
|
||||
"""OAuth provider should be used for sk-ant-oat tokens."""
|
||||
assert should_use_oauth_provider("sk-ant-oat01-xxx", "anthropic/claude-opus-4-7") is True
|
||||
assert should_use_oauth_provider("sk-ant-oat01-xxx", "claude-sonnet-4") is True
|
||||
|
||||
|
||||
def test_should_not_use_oauth_for_regular_key():
|
||||
"""Regular API keys should not use OAuth provider."""
|
||||
assert should_use_oauth_provider("sk-ant-api03-xxx", "claude-opus-4-7") is False
|
||||
assert should_use_oauth_provider("sk-or-v1-xxx", "anthropic/claude-opus-4-7") is False
|
||||
|
||||
|
||||
def test_should_not_use_oauth_for_non_anthropic():
|
||||
"""Non-Anthropic models should not use OAuth provider."""
|
||||
assert should_use_oauth_provider("sk-ant-oat01-xxx", "gpt-4") is False
|
||||
assert should_use_oauth_provider("sk-ant-oat01-xxx", "deepseek/deepseek-chat") is False
|
||||
@@ -0,0 +1,137 @@
|
||||
"""Test SessionManager audit log functionality."""
|
||||
|
||||
import json
|
||||
|
||||
import pytest
|
||||
|
||||
from nanobot.session.manager import Session, SessionManager
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def session_manager(tmp_path):
|
||||
return SessionManager(workspace=tmp_path)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def session():
|
||||
s = Session(key="telegram:12345")
|
||||
s.add_message("user", "Hello")
|
||||
s.add_message("assistant", "Hi there!")
|
||||
return s
|
||||
|
||||
|
||||
def test_save_creates_audit_file(session_manager, session):
|
||||
"""SessionManager.save() should create a monthly audit log file."""
|
||||
session_manager.save(session)
|
||||
|
||||
audit_files = list(session_manager.sessions_dir.glob("*.audit.*.jsonl"))
|
||||
assert len(audit_files) == 1
|
||||
assert "telegram_12345.audit." in audit_files[0].name
|
||||
|
||||
|
||||
def test_audit_file_contains_save_marker(session_manager, session):
|
||||
"""Audit log should start with a save_marker line containing metadata."""
|
||||
session_manager.save(session)
|
||||
|
||||
audit_files = list(session_manager.sessions_dir.glob("*.audit.*.jsonl"))
|
||||
lines = audit_files[0].read_text().strip().split("\n")
|
||||
|
||||
marker = json.loads(lines[0])
|
||||
assert marker["_type"] == "save_marker"
|
||||
assert marker["message_count"] == 2
|
||||
assert "timestamp" in marker
|
||||
|
||||
|
||||
def test_audit_file_contains_all_messages(session_manager, session):
|
||||
"""Audit log should contain all session messages after the save marker."""
|
||||
session_manager.save(session)
|
||||
|
||||
audit_files = list(session_manager.sessions_dir.glob("*.audit.*.jsonl"))
|
||||
lines = audit_files[0].read_text().strip().split("\n")
|
||||
|
||||
# Line 0 = save_marker, lines 1-2 = messages
|
||||
assert len(lines) == 3
|
||||
msg1 = json.loads(lines[1])
|
||||
msg2 = json.loads(lines[2])
|
||||
assert msg1["role"] == "user"
|
||||
assert msg1["content"] == "Hello"
|
||||
assert msg2["role"] == "assistant"
|
||||
assert msg2["content"] == "Hi there!"
|
||||
|
||||
|
||||
def test_audit_file_is_append_only(session_manager, session):
|
||||
"""Multiple saves should append to the same audit file, not overwrite."""
|
||||
session_manager.save(session)
|
||||
|
||||
# Add another message and save again
|
||||
session.add_message("user", "How are you?")
|
||||
session_manager.save(session)
|
||||
|
||||
audit_files = list(session_manager.sessions_dir.glob("*.audit.*.jsonl"))
|
||||
assert len(audit_files) == 1 # Same file
|
||||
|
||||
lines = audit_files[0].read_text().strip().split("\n")
|
||||
|
||||
# First save: 1 marker + 2 messages = 3 lines
|
||||
# Second save: 1 marker + 3 messages = 4 lines
|
||||
# Total: 7 lines
|
||||
assert len(lines) == 7
|
||||
|
||||
# Both save markers present
|
||||
markers = [json.loads(l) for l in lines if json.loads(l).get("_type") == "save_marker"]
|
||||
assert len(markers) == 2
|
||||
assert markers[0]["message_count"] == 2
|
||||
assert markers[1]["message_count"] == 3
|
||||
|
||||
|
||||
def test_audit_preserves_message_fields(session_manager):
|
||||
"""Audit log should preserve all message fields including reasoning_content."""
|
||||
session = Session(key="test:preserve")
|
||||
session.add_raw_message({
|
||||
"role": "assistant",
|
||||
"content": "thinking response",
|
||||
"reasoning_content": [{"type": "thinking", "thinking": "deep thoughts"}],
|
||||
"timestamp": "2026-03-22T12:00:00",
|
||||
})
|
||||
|
||||
session_manager.save(session)
|
||||
|
||||
audit_files = list(session_manager.sessions_dir.glob("*.audit.*.jsonl"))
|
||||
lines = audit_files[0].read_text().strip().split("\n")
|
||||
|
||||
msg = json.loads(lines[1])
|
||||
assert msg["reasoning_content"] == [{"type": "thinking", "thinking": "deep thoughts"}]
|
||||
|
||||
|
||||
def test_audit_failure_does_not_break_save(session_manager, session, tmp_path):
|
||||
"""If audit logging fails, the main session save should still succeed.
|
||||
|
||||
_append_audit has its own try/except, so internal failures are caught.
|
||||
We simulate a realistic failure by making the sessions dir read-only
|
||||
for audit file creation.
|
||||
"""
|
||||
# First save works (creates both session file and audit file)
|
||||
session_manager.save(session)
|
||||
|
||||
path = session_manager._get_session_path(session.key)
|
||||
assert path.exists()
|
||||
|
||||
# Remove audit files and make a blocking file at the audit path
|
||||
# so the next audit open("a") fails
|
||||
for af in session_manager.sessions_dir.glob("*.audit.*.jsonl"):
|
||||
af.unlink()
|
||||
|
||||
# Create a directory where the audit file should be — open() will fail
|
||||
from datetime import datetime
|
||||
now = datetime.now()
|
||||
bad_path = session_manager.sessions_dir / f"telegram_12345.audit.{now:%Y-%m}.jsonl"
|
||||
bad_path.mkdir()
|
||||
|
||||
# Second save should succeed despite audit failure
|
||||
session.add_message("user", "another message")
|
||||
session_manager.save(session)
|
||||
|
||||
# Session file should still be written correctly
|
||||
with open(path) as f:
|
||||
first_line = json.loads(f.readline())
|
||||
assert first_line["_type"] == "metadata"
|
||||
@@ -0,0 +1,139 @@
|
||||
# tests/test_subagent_wait.py
|
||||
"""Tests for wait_for_subagents with top-level and child subagents."""
|
||||
import pytest
|
||||
from pathlib import Path
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
from nanobot.agent.subagent import SubagentManager
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.providers.base import LLMProvider, LLMResponse
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_wait_for_top_level_subagent():
|
||||
"""Test that wait_for works for top-level subagents spawned from telegram."""
|
||||
bus = MessageBus()
|
||||
provider = MagicMock(spec=LLMProvider)
|
||||
provider.chat = AsyncMock(return_value=LLMResponse(
|
||||
content="Task completed",
|
||||
tool_calls=[]
|
||||
))
|
||||
provider.get_default_model = MagicMock(return_value="test-model")
|
||||
provider.thinking_budget = 0
|
||||
|
||||
workspace = Path("/tmp/test-subagent")
|
||||
workspace.mkdir(exist_ok=True)
|
||||
|
||||
manager = SubagentManager(
|
||||
bus=bus,
|
||||
provider=provider,
|
||||
workspace=workspace
|
||||
)
|
||||
|
||||
# Spawn a top-level subagent (origin channel = "telegram")
|
||||
task_id = await manager.spawn(
|
||||
task="Test task",
|
||||
label="test",
|
||||
model=None,
|
||||
origin_channel="telegram",
|
||||
origin_chat_id="12345"
|
||||
)
|
||||
|
||||
# Wait for it to complete
|
||||
result = await manager.wait_for([task_id])
|
||||
|
||||
# Should find the result (not "No result found")
|
||||
assert "No result found" not in result
|
||||
assert task_id in result
|
||||
assert "Task completed" in result or "completed" in result.lower()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_wait_for_child_subagent():
|
||||
"""Test that wait_for works for child subagents (orchestrator pattern)."""
|
||||
bus = MessageBus()
|
||||
provider = MagicMock(spec=LLMProvider)
|
||||
provider.chat = AsyncMock(return_value=LLMResponse(
|
||||
content="Child task completed",
|
||||
tool_calls=[]
|
||||
))
|
||||
provider.get_default_model = MagicMock(return_value="test-model")
|
||||
provider.thinking_budget = 0
|
||||
|
||||
workspace = Path("/tmp/test-subagent")
|
||||
workspace.mkdir(exist_ok=True)
|
||||
|
||||
manager = SubagentManager(
|
||||
bus=bus,
|
||||
provider=provider,
|
||||
workspace=workspace
|
||||
)
|
||||
|
||||
# Spawn a child subagent (origin channel = "subagent")
|
||||
task_id = await manager.spawn(
|
||||
task="Child test task",
|
||||
label="test-child",
|
||||
model=None,
|
||||
origin_channel="subagent",
|
||||
origin_chat_id="parent-id"
|
||||
)
|
||||
|
||||
# Wait for it to complete
|
||||
result = await manager.wait_for([task_id])
|
||||
|
||||
# Should find the result (not "No result found")
|
||||
assert "No result found" not in result
|
||||
assert task_id in result
|
||||
assert "Child task completed" in result or "completed" in result.lower()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_wait_for_multiple_subagents():
|
||||
"""Test waiting for multiple subagents of different types."""
|
||||
bus = MessageBus()
|
||||
call_count = 0
|
||||
|
||||
async def chat_response(*args, **kwargs):
|
||||
nonlocal call_count
|
||||
call_count += 1
|
||||
return LLMResponse(content=f"Task {call_count} completed", tool_calls=[])
|
||||
|
||||
provider = MagicMock(spec=LLMProvider)
|
||||
provider.chat = AsyncMock(side_effect=chat_response)
|
||||
provider.get_default_model = MagicMock(return_value="test-model")
|
||||
provider.thinking_budget = 0
|
||||
|
||||
workspace = Path("/tmp/test-subagent")
|
||||
workspace.mkdir(exist_ok=True)
|
||||
|
||||
manager = SubagentManager(
|
||||
bus=bus,
|
||||
provider=provider,
|
||||
workspace=workspace
|
||||
)
|
||||
|
||||
# Spawn one top-level and one child subagent
|
||||
task_id_1 = await manager.spawn(
|
||||
task="Top-level task",
|
||||
label="test-top",
|
||||
model=None,
|
||||
origin_channel="telegram",
|
||||
origin_chat_id="12345"
|
||||
)
|
||||
|
||||
task_id_2 = await manager.spawn(
|
||||
task="Child task",
|
||||
label="test-child",
|
||||
model=None,
|
||||
origin_channel="subagent",
|
||||
origin_chat_id="parent"
|
||||
)
|
||||
|
||||
# Wait for both
|
||||
result = await manager.wait_for([task_id_1, task_id_2])
|
||||
|
||||
# Should find both results
|
||||
assert "No result found" not in result
|
||||
assert task_id_1 in result
|
||||
assert task_id_2 in result
|
||||
assert "Task 1 completed" in result or "completed" in result.lower()
|
||||
assert "Task 2 completed" in result or "completed" in result.lower()
|
||||
@@ -1,167 +0,0 @@
|
||||
"""Tests for /stop task cancellation."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
def _make_loop():
|
||||
"""Create a minimal AgentLoop with mocked dependencies."""
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.queue import MessageBus
|
||||
|
||||
bus = MessageBus()
|
||||
provider = MagicMock()
|
||||
provider.get_default_model.return_value = "test-model"
|
||||
workspace = MagicMock()
|
||||
workspace.__truediv__ = MagicMock(return_value=MagicMock())
|
||||
|
||||
with patch("nanobot.agent.loop.ContextBuilder"), \
|
||||
patch("nanobot.agent.loop.SessionManager"), \
|
||||
patch("nanobot.agent.loop.SubagentManager") as MockSubMgr:
|
||||
MockSubMgr.return_value.cancel_by_session = AsyncMock(return_value=0)
|
||||
loop = AgentLoop(bus=bus, provider=provider, workspace=workspace)
|
||||
return loop, bus
|
||||
|
||||
|
||||
class TestHandleStop:
|
||||
@pytest.mark.asyncio
|
||||
async def test_stop_no_active_task(self):
|
||||
from nanobot.bus.events import InboundMessage
|
||||
|
||||
loop, bus = _make_loop()
|
||||
msg = InboundMessage(channel="test", sender_id="u1", chat_id="c1", content="/stop")
|
||||
await loop._handle_stop(msg)
|
||||
out = await asyncio.wait_for(bus.consume_outbound(), timeout=1.0)
|
||||
assert "No active task" in out.content
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_stop_cancels_active_task(self):
|
||||
from nanobot.bus.events import InboundMessage
|
||||
|
||||
loop, bus = _make_loop()
|
||||
cancelled = asyncio.Event()
|
||||
|
||||
async def slow_task():
|
||||
try:
|
||||
await asyncio.sleep(60)
|
||||
except asyncio.CancelledError:
|
||||
cancelled.set()
|
||||
raise
|
||||
|
||||
task = asyncio.create_task(slow_task())
|
||||
await asyncio.sleep(0)
|
||||
loop._active_tasks["test:c1"] = [task]
|
||||
|
||||
msg = InboundMessage(channel="test", sender_id="u1", chat_id="c1", content="/stop")
|
||||
await loop._handle_stop(msg)
|
||||
|
||||
assert cancelled.is_set()
|
||||
out = await asyncio.wait_for(bus.consume_outbound(), timeout=1.0)
|
||||
assert "stopped" in out.content.lower()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_stop_cancels_multiple_tasks(self):
|
||||
from nanobot.bus.events import InboundMessage
|
||||
|
||||
loop, bus = _make_loop()
|
||||
events = [asyncio.Event(), asyncio.Event()]
|
||||
|
||||
async def slow(idx):
|
||||
try:
|
||||
await asyncio.sleep(60)
|
||||
except asyncio.CancelledError:
|
||||
events[idx].set()
|
||||
raise
|
||||
|
||||
tasks = [asyncio.create_task(slow(i)) for i in range(2)]
|
||||
await asyncio.sleep(0)
|
||||
loop._active_tasks["test:c1"] = tasks
|
||||
|
||||
msg = InboundMessage(channel="test", sender_id="u1", chat_id="c1", content="/stop")
|
||||
await loop._handle_stop(msg)
|
||||
|
||||
assert all(e.is_set() for e in events)
|
||||
out = await asyncio.wait_for(bus.consume_outbound(), timeout=1.0)
|
||||
assert "2 task" in out.content
|
||||
|
||||
|
||||
class TestDispatch:
|
||||
@pytest.mark.asyncio
|
||||
async def test_dispatch_processes_and_publishes(self):
|
||||
from nanobot.bus.events import InboundMessage, OutboundMessage
|
||||
|
||||
loop, bus = _make_loop()
|
||||
msg = InboundMessage(channel="test", sender_id="u1", chat_id="c1", content="hello")
|
||||
loop._process_message = AsyncMock(
|
||||
return_value=OutboundMessage(channel="test", chat_id="c1", content="hi")
|
||||
)
|
||||
await loop._dispatch(msg)
|
||||
out = await asyncio.wait_for(bus.consume_outbound(), timeout=1.0)
|
||||
assert out.content == "hi"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_processing_lock_serializes(self):
|
||||
from nanobot.bus.events import InboundMessage, OutboundMessage
|
||||
|
||||
loop, bus = _make_loop()
|
||||
order = []
|
||||
|
||||
async def mock_process(m, **kwargs):
|
||||
order.append(f"start-{m.content}")
|
||||
await asyncio.sleep(0.05)
|
||||
order.append(f"end-{m.content}")
|
||||
return OutboundMessage(channel="test", chat_id="c1", content=m.content)
|
||||
|
||||
loop._process_message = mock_process
|
||||
msg1 = InboundMessage(channel="test", sender_id="u1", chat_id="c1", content="a")
|
||||
msg2 = InboundMessage(channel="test", sender_id="u1", chat_id="c1", content="b")
|
||||
|
||||
t1 = asyncio.create_task(loop._dispatch(msg1))
|
||||
t2 = asyncio.create_task(loop._dispatch(msg2))
|
||||
await asyncio.gather(t1, t2)
|
||||
assert order == ["start-a", "end-a", "start-b", "end-b"]
|
||||
|
||||
|
||||
class TestSubagentCancellation:
|
||||
@pytest.mark.asyncio
|
||||
async def test_cancel_by_session(self):
|
||||
from nanobot.agent.subagent import SubagentManager
|
||||
from nanobot.bus.queue import MessageBus
|
||||
|
||||
bus = MessageBus()
|
||||
provider = MagicMock()
|
||||
provider.get_default_model.return_value = "test-model"
|
||||
mgr = SubagentManager(provider=provider, workspace=MagicMock(), bus=bus)
|
||||
|
||||
cancelled = asyncio.Event()
|
||||
|
||||
async def slow():
|
||||
try:
|
||||
await asyncio.sleep(60)
|
||||
except asyncio.CancelledError:
|
||||
cancelled.set()
|
||||
raise
|
||||
|
||||
task = asyncio.create_task(slow())
|
||||
await asyncio.sleep(0)
|
||||
mgr._running_tasks["sub-1"] = task
|
||||
mgr._session_tasks["test:c1"] = {"sub-1"}
|
||||
|
||||
count = await mgr.cancel_by_session("test:c1")
|
||||
assert count == 1
|
||||
assert cancelled.is_set()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_cancel_by_session_no_tasks(self):
|
||||
from nanobot.agent.subagent import SubagentManager
|
||||
from nanobot.bus.queue import MessageBus
|
||||
|
||||
bus = MessageBus()
|
||||
provider = MagicMock()
|
||||
provider.get_default_model.return_value = "test-model"
|
||||
mgr = SubagentManager(provider=provider, workspace=MagicMock(), bus=bus)
|
||||
assert await mgr.cancel_by_session("nonexistent") == 0
|
||||
@@ -0,0 +1,101 @@
|
||||
"""Integration tests for Telegram media sending."""
|
||||
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from nanobot.bus.events import OutboundMessage
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.channels.telegram import TelegramChannel
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def mock_telegram_app():
|
||||
"""Mock python-telegram-bot Application."""
|
||||
app = MagicMock()
|
||||
app.bot = MagicMock()
|
||||
app.bot.send_photo = AsyncMock()
|
||||
app.bot.send_video = AsyncMock()
|
||||
app.bot.send_audio = AsyncMock()
|
||||
app.bot.send_document = AsyncMock()
|
||||
app.bot.send_media_group = AsyncMock()
|
||||
app.bot.send_message = AsyncMock()
|
||||
return app
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_send_single_image(mock_telegram_app, tmp_path):
|
||||
"""Test sending single image."""
|
||||
# Create test image
|
||||
from PIL import Image
|
||||
img_path = tmp_path / "test.jpg"
|
||||
img = Image.new("RGB", (100, 100), color="red")
|
||||
img.save(img_path, format="JPEG")
|
||||
|
||||
# Setup channel
|
||||
bus = MessageBus()
|
||||
config = MagicMock()
|
||||
config.token = "fake_token"
|
||||
config.proxy = None
|
||||
|
||||
channel = TelegramChannel(config, bus)
|
||||
channel._app = mock_telegram_app
|
||||
channel._running = True
|
||||
|
||||
# Send message with media
|
||||
msg = OutboundMessage(
|
||||
channel="telegram",
|
||||
chat_id="12345",
|
||||
content="Test image",
|
||||
media=[str(img_path)]
|
||||
)
|
||||
|
||||
with patch("nanobot.channels.telegram._markdown_to_telegram_html", return_value="Test image"):
|
||||
await channel.send(msg)
|
||||
|
||||
# Verify send_photo was called
|
||||
mock_telegram_app.bot.send_photo.assert_called_once()
|
||||
call_args = mock_telegram_app.bot.send_photo.call_args
|
||||
assert call_args.kwargs["chat_id"] == 12345
|
||||
assert call_args.kwargs["caption"] == "Test image"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_send_album(mock_telegram_app, tmp_path):
|
||||
"""Test sending multiple images as album."""
|
||||
from PIL import Image
|
||||
|
||||
# Create test images
|
||||
img_paths = []
|
||||
for i in range(3):
|
||||
img_path = tmp_path / f"test{i}.jpg"
|
||||
img = Image.new("RGB", (100, 100), color="red")
|
||||
img.save(img_path, format="JPEG")
|
||||
img_paths.append(str(img_path))
|
||||
|
||||
# Setup channel
|
||||
bus = MessageBus()
|
||||
config = MagicMock()
|
||||
config.token = "fake_token"
|
||||
config.proxy = None
|
||||
|
||||
channel = TelegramChannel(config, bus)
|
||||
channel._app = mock_telegram_app
|
||||
channel._running = True
|
||||
|
||||
# Send message with multiple images
|
||||
msg = OutboundMessage(
|
||||
channel="telegram",
|
||||
chat_id="12345",
|
||||
content="Album test",
|
||||
media=img_paths
|
||||
)
|
||||
|
||||
with patch("nanobot.channels.telegram._markdown_to_telegram_html", return_value="Album test"):
|
||||
await channel.send(msg)
|
||||
|
||||
# Verify send_media_group was called
|
||||
mock_telegram_app.bot.send_media_group.assert_called_once()
|
||||
call_args = mock_telegram_app.bot.send_media_group.call_args
|
||||
assert call_args.kwargs["chat_id"] == 12345
|
||||
assert len(call_args.kwargs["media"]) == 3
|
||||
@@ -0,0 +1,170 @@
|
||||
"""Tests for Telegram message chunking.
|
||||
|
||||
Per design doc: messages >4096 chars should split at sentence boundaries.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
from nanobot.bus.events import OutboundMessage
|
||||
from nanobot.channels.telegram import TelegramChannel
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_short_message_not_chunked():
|
||||
"""Messages under 4096 chars should send as single message."""
|
||||
config = MagicMock()
|
||||
config.token = "test-token"
|
||||
bus = MagicMock()
|
||||
|
||||
channel = TelegramChannel(config, bus)
|
||||
|
||||
# Mock the bot
|
||||
sent_messages = []
|
||||
|
||||
class MockBot:
|
||||
async def send_message(self, chat_id, text, parse_mode=None):
|
||||
sent_messages.append({"chat_id": chat_id, "text": text})
|
||||
|
||||
class MockApp:
|
||||
bot = MockBot()
|
||||
|
||||
channel._app = MockApp()
|
||||
|
||||
# Short message
|
||||
msg = OutboundMessage(
|
||||
channel="telegram",
|
||||
chat_id="123",
|
||||
content="Short message."
|
||||
)
|
||||
|
||||
await channel.send(msg)
|
||||
|
||||
# Should send exactly 1 message
|
||||
assert len(sent_messages) == 1
|
||||
assert sent_messages[0]["text"] == "Short message."
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_long_message_splits_at_sentences():
|
||||
"""Messages >4096 chars should split at sentence boundaries."""
|
||||
config = MagicMock()
|
||||
config.token = "test-token"
|
||||
bus = MagicMock()
|
||||
|
||||
channel = TelegramChannel(config, bus)
|
||||
|
||||
# Mock the bot
|
||||
sent_messages = []
|
||||
|
||||
class MockBot:
|
||||
async def send_message(self, chat_id, text, parse_mode=None):
|
||||
sent_messages.append({"chat_id": chat_id, "text": text})
|
||||
|
||||
class MockApp:
|
||||
bot = MockBot()
|
||||
|
||||
channel._app = MockApp()
|
||||
|
||||
# Create a message longer than 4096 chars with clear sentence boundaries
|
||||
# Each sentence is 200 chars, need 21+ sentences to exceed 4096
|
||||
sentence = "A" * 195 + "end. " # 200 chars including "end. "
|
||||
long_content = sentence * 25 # 5000 chars total
|
||||
|
||||
msg = OutboundMessage(
|
||||
channel="telegram",
|
||||
chat_id="123",
|
||||
content=long_content
|
||||
)
|
||||
|
||||
await channel.send(msg)
|
||||
|
||||
# Should split into multiple messages
|
||||
assert len(sent_messages) > 1
|
||||
|
||||
# Each message should be under 4096 chars
|
||||
for sent in sent_messages:
|
||||
assert len(sent["text"]) <= 4096
|
||||
|
||||
# All messages combined should equal original (with whitespace trimming)
|
||||
combined = "".join(sent["text"] for sent in sent_messages)
|
||||
assert combined.replace(" ", "") == long_content.replace(" ", "")
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_message_at_exactly_4000_chars():
|
||||
"""Message at exactly 4000 chars should not chunk (safer limit)."""
|
||||
config = MagicMock()
|
||||
config.token = "test-token"
|
||||
bus = MagicMock()
|
||||
|
||||
channel = TelegramChannel(config, bus)
|
||||
|
||||
# Mock the bot
|
||||
sent_messages = []
|
||||
|
||||
class MockBot:
|
||||
async def send_message(self, chat_id, text, parse_mode=None):
|
||||
sent_messages.append({"chat_id": chat_id, "text": text})
|
||||
|
||||
class MockApp:
|
||||
bot = MockBot()
|
||||
|
||||
channel._app = MockApp()
|
||||
|
||||
# Exactly 4000 chars (upstream uses 4000 as safer limit vs 4096)
|
||||
content = "A" * 4000
|
||||
|
||||
msg = OutboundMessage(
|
||||
channel="telegram",
|
||||
chat_id="123",
|
||||
content=content
|
||||
)
|
||||
|
||||
await channel.send(msg)
|
||||
|
||||
# Should send exactly 1 message
|
||||
assert len(sent_messages) == 1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_message_preserves_sentence_boundaries():
|
||||
"""Chunks should split at sentence endings, not mid-sentence."""
|
||||
config = MagicMock()
|
||||
config.token = "test-token"
|
||||
bus = MagicMock()
|
||||
|
||||
channel = TelegramChannel(config, bus)
|
||||
|
||||
# Mock the bot
|
||||
sent_messages = []
|
||||
|
||||
class MockBot:
|
||||
async def send_message(self, chat_id, text, parse_mode=None):
|
||||
sent_messages.append({"chat_id": chat_id, "text": text})
|
||||
|
||||
class MockApp:
|
||||
bot = MockBot()
|
||||
|
||||
channel._app = MockApp()
|
||||
|
||||
# Create content with clear sentence markers
|
||||
# First part: just under 4096 chars
|
||||
part1 = "First sentence. " * 250 # ~4000 chars
|
||||
part2 = "Second sentence. "
|
||||
content = part1 + part2
|
||||
|
||||
msg = OutboundMessage(
|
||||
channel="telegram",
|
||||
chat_id="123",
|
||||
content=content
|
||||
)
|
||||
|
||||
await channel.send(msg)
|
||||
|
||||
# Verify chunks don't break mid-sentence
|
||||
for sent in sent_messages:
|
||||
text = sent["text"].strip()
|
||||
# Each chunk should end with sentence punctuation
|
||||
if text:
|
||||
assert text[-1] in ".!?"
|
||||
@@ -0,0 +1,263 @@
|
||||
"""Tests for Telegram media handling."""
|
||||
|
||||
import io
|
||||
|
||||
import pytest
|
||||
from PIL import Image
|
||||
|
||||
|
||||
def test_detect_mime_from_jpeg():
|
||||
"""Test MIME detection for JPEG images."""
|
||||
from nanobot.channels.telegram_media import detect_mime
|
||||
|
||||
# Create minimal JPEG bytes (FF D8 FF = JPEG magic bytes)
|
||||
jpeg_bytes = b'\xff\xd8\xff\xe0\x00\x10JFIF'
|
||||
|
||||
mime = detect_mime("test.jpg", jpeg_bytes)
|
||||
assert mime == "image/jpeg"
|
||||
|
||||
|
||||
def test_detect_mime_from_png():
|
||||
"""Test MIME detection for PNG images."""
|
||||
from nanobot.channels.telegram_media import detect_mime
|
||||
|
||||
# PNG magic bytes
|
||||
png_bytes = b'\x89PNG\r\n\x1a\n'
|
||||
|
||||
mime = detect_mime("test.png", png_bytes)
|
||||
assert mime == "image/png"
|
||||
|
||||
|
||||
def test_detect_mime_from_extension_fallback():
|
||||
"""Test MIME detection falls back to extension when no content provided."""
|
||||
from nanobot.channels.telegram_media import detect_mime
|
||||
|
||||
mime = detect_mime("video.mp4", None)
|
||||
assert mime == "video/mp4"
|
||||
|
||||
|
||||
def test_detect_mime_unknown():
|
||||
"""Test MIME detection returns generic type for unknown files."""
|
||||
from nanobot.channels.telegram_media import detect_mime
|
||||
|
||||
mime = detect_mime("unknown.xyz", None)
|
||||
assert mime == "application/octet-stream"
|
||||
|
||||
|
||||
def test_detect_mime_magic_fallback_on_octet_stream():
|
||||
"""Test that extension is preferred when magic returns generic type."""
|
||||
from nanobot.channels.telegram_media import detect_mime
|
||||
|
||||
# Generic binary content that magic might identify as octet-stream
|
||||
generic_bytes = b'\x00\x01\x02\x03'
|
||||
|
||||
# But extension clearly indicates it's an image
|
||||
mime = detect_mime("image.png", generic_bytes)
|
||||
|
||||
# Should use extension (png) not magic's generic result
|
||||
# Note: This tests the logic at line 36 - avoiding generic types
|
||||
assert mime in ("image/png", "application/octet-stream")
|
||||
|
||||
|
||||
def test_detect_mime_malformed_content():
|
||||
"""Test fallback when magic detection fails with malformed content."""
|
||||
from nanobot.channels.telegram_media import detect_mime
|
||||
|
||||
# Malformed content that might cause magic to raise an exception
|
||||
malformed = b'\xff' * 10
|
||||
|
||||
# Should fallback to extension detection, not crash
|
||||
mime = detect_mime("test.mp4", malformed)
|
||||
assert mime == "video/mp4"
|
||||
|
||||
|
||||
def test_classify_media_image():
|
||||
"""Test classification of image MIME types."""
|
||||
from nanobot.channels.telegram_media import MediaKind, classify_media
|
||||
|
||||
assert classify_media("image/jpeg") == MediaKind.IMAGE
|
||||
assert classify_media("image/png") == MediaKind.IMAGE
|
||||
assert classify_media("image/webp") == MediaKind.IMAGE
|
||||
|
||||
|
||||
def test_classify_media_video():
|
||||
"""Test classification of video MIME types."""
|
||||
from nanobot.channels.telegram_media import MediaKind, classify_media
|
||||
|
||||
assert classify_media("video/mp4") == MediaKind.VIDEO
|
||||
assert classify_media("video/quicktime") == MediaKind.VIDEO
|
||||
|
||||
|
||||
def test_classify_media_audio():
|
||||
"""Test classification of audio MIME types."""
|
||||
from nanobot.channels.telegram_media import MediaKind, classify_media
|
||||
|
||||
assert classify_media("audio/mpeg") == MediaKind.AUDIO
|
||||
assert classify_media("audio/ogg") == MediaKind.AUDIO
|
||||
|
||||
|
||||
def test_classify_media_document():
|
||||
"""Test classification of document MIME types."""
|
||||
from nanobot.channels.telegram_media import MediaKind, classify_media
|
||||
|
||||
assert classify_media("application/pdf") == MediaKind.DOCUMENT
|
||||
assert classify_media("text/plain") == MediaKind.DOCUMENT
|
||||
assert classify_media("application/octet-stream") == MediaKind.DOCUMENT
|
||||
|
||||
|
||||
def test_is_heic_format():
|
||||
"""Test HEIC format detection."""
|
||||
from nanobot.channels.telegram_media import is_heic_format
|
||||
|
||||
assert is_heic_format("photo.heic") is True
|
||||
assert is_heic_format("photo.HEIC") is True
|
||||
assert is_heic_format("photo.heif") is True
|
||||
assert is_heic_format("photo.jpg") is False
|
||||
|
||||
|
||||
def test_optimize_image_jpeg_quality(tmp_path):
|
||||
"""Test JPEG optimization reduces size with quality ladder."""
|
||||
from nanobot.channels.telegram_media import optimize_image
|
||||
|
||||
# Create a large test image (3000x3000 RGB)
|
||||
img = Image.new("RGB", (3000, 3000), color="red")
|
||||
buf = io.BytesIO()
|
||||
img.save(buf, format="JPEG", quality=95)
|
||||
original_bytes = buf.getvalue()
|
||||
original_size = len(original_bytes)
|
||||
|
||||
# Write to temp file
|
||||
temp_file = tmp_path / "test.jpg"
|
||||
temp_file.write_bytes(original_bytes)
|
||||
|
||||
# Optimize to 1MB max
|
||||
optimized = optimize_image(str(temp_file), max_bytes=1_000_000)
|
||||
|
||||
# Should be smaller than original
|
||||
assert len(optimized) < original_size
|
||||
# Should be under limit
|
||||
assert len(optimized) <= 1_000_000
|
||||
# Should still be valid JPEG
|
||||
assert optimized.startswith(b'\xff\xd8\xff')
|
||||
|
||||
|
||||
def test_optimize_image_png_preserve_alpha(tmp_path):
|
||||
"""Test PNG with alpha channel is preserved."""
|
||||
from nanobot.channels.telegram_media import optimize_image
|
||||
|
||||
# Create PNG with alpha channel
|
||||
img = Image.new("RGBA", (1000, 1000), color=(255, 0, 0, 128))
|
||||
buf = io.BytesIO()
|
||||
img.save(buf, format="PNG")
|
||||
original_bytes = buf.getvalue()
|
||||
|
||||
# Write to temp file
|
||||
temp_file = tmp_path / "test.png"
|
||||
temp_file.write_bytes(original_bytes)
|
||||
|
||||
optimized = optimize_image(str(temp_file), max_bytes=5_000_000)
|
||||
|
||||
# Should still be PNG (PNG magic bytes)
|
||||
assert optimized.startswith(b'\x89PNG')
|
||||
|
||||
# Load and verify alpha channel preserved
|
||||
img_opt = Image.open(io.BytesIO(optimized))
|
||||
assert img_opt.mode == "RGBA"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_fetch_media_success():
|
||||
"""Test fetching media from remote URL."""
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
from nanobot.channels.telegram_media import fetch_media
|
||||
|
||||
# Mock httpx response
|
||||
mock_content = b"fake image data"
|
||||
mock_response = MagicMock()
|
||||
mock_response.content = mock_content
|
||||
mock_response.headers = {"content-type": "image/jpeg"}
|
||||
mock_response.raise_for_status = MagicMock()
|
||||
|
||||
with patch("httpx.AsyncClient") as mock_client:
|
||||
mock_client.return_value.__aenter__.return_value.get = AsyncMock(return_value=mock_response)
|
||||
|
||||
content, mime = await fetch_media("https://example.com/image.jpg", max_bytes=10_000_000)
|
||||
|
||||
assert content == mock_content
|
||||
assert mime == "image/jpeg"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_fetch_media_timeout():
|
||||
"""Test fetch media handles timeout."""
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
import httpx
|
||||
|
||||
from nanobot.channels.telegram_media import fetch_media
|
||||
|
||||
with patch("httpx.AsyncClient") as mock_client:
|
||||
mock_client.return_value.__aenter__.return_value.get = AsyncMock(side_effect=httpx.TimeoutException("timeout"))
|
||||
|
||||
with pytest.raises(ValueError, match="timeout"):
|
||||
await fetch_media("https://example.com/image.jpg", max_bytes=10_000_000)
|
||||
|
||||
|
||||
def test_group_media_all_images():
|
||||
"""Test grouping all images into album."""
|
||||
from nanobot.channels.telegram_media import MediaKind, group_media_for_album
|
||||
|
||||
media_items = [
|
||||
("image1.jpg", MediaKind.IMAGE),
|
||||
("image2.png", MediaKind.IMAGE),
|
||||
("image3.jpeg", MediaKind.IMAGE),
|
||||
]
|
||||
|
||||
result = group_media_for_album(media_items)
|
||||
|
||||
assert result["album"] == ["image1.jpg", "image2.png", "image3.jpeg"]
|
||||
assert result["separate"] == []
|
||||
|
||||
|
||||
def test_group_media_all_videos():
|
||||
"""Test grouping all videos into album."""
|
||||
from nanobot.channels.telegram_media import MediaKind, group_media_for_album
|
||||
|
||||
media_items = [
|
||||
("video1.mp4", MediaKind.VIDEO),
|
||||
("video2.mov", MediaKind.VIDEO),
|
||||
]
|
||||
|
||||
result = group_media_for_album(media_items)
|
||||
|
||||
assert result["album"] == ["video1.mp4", "video2.mov"]
|
||||
assert result["separate"] == []
|
||||
|
||||
|
||||
def test_group_media_mixed_types():
|
||||
"""Test mixed media types sent separately."""
|
||||
from nanobot.channels.telegram_media import MediaKind, group_media_for_album
|
||||
|
||||
media_items = [
|
||||
("image.jpg", MediaKind.IMAGE),
|
||||
("video.mp4", MediaKind.VIDEO),
|
||||
("audio.mp3", MediaKind.AUDIO),
|
||||
]
|
||||
|
||||
result = group_media_for_album(media_items)
|
||||
|
||||
assert result["album"] == []
|
||||
assert result["separate"] == ["image.jpg", "video.mp4", "audio.mp3"]
|
||||
|
||||
|
||||
def test_group_media_single_item():
|
||||
"""Test single media item sent separately (not as album)."""
|
||||
from nanobot.channels.telegram_media import MediaKind, group_media_for_album
|
||||
|
||||
media_items = [("image.jpg", MediaKind.IMAGE)]
|
||||
|
||||
result = group_media_for_album(media_items)
|
||||
|
||||
assert result["album"] == []
|
||||
assert result["separate"] == ["image.jpg"]
|
||||
@@ -0,0 +1,65 @@
|
||||
# tests/test_telegram_suppress.py
|
||||
import pytest
|
||||
from nanobot.bus.events import OutboundMessage
|
||||
from nanobot.channels.telegram import TelegramChannel
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_suppressed_message_not_sent():
|
||||
"""Test that messages with suppressed=True metadata are not sent to Telegram API."""
|
||||
|
||||
config = MagicMock()
|
||||
config.token = "test-token"
|
||||
bus = MagicMock()
|
||||
|
||||
channel = TelegramChannel(config, bus)
|
||||
|
||||
# Mock the internal _app and bot directly (skip start())
|
||||
mock_app = MagicMock()
|
||||
mock_bot = AsyncMock()
|
||||
mock_app.bot = mock_bot
|
||||
channel._app = mock_app
|
||||
|
||||
# Send a suppressed message
|
||||
msg = OutboundMessage(
|
||||
channel="telegram",
|
||||
chat_id="12345",
|
||||
content="[HIDDEN] This should not be sent",
|
||||
metadata={"suppressed": True}
|
||||
)
|
||||
|
||||
await channel.send(msg)
|
||||
|
||||
# Verify bot.send_message was NOT called
|
||||
mock_bot.send_message.assert_not_called()
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_normal_message_sent():
|
||||
"""Test that normal messages are sent to Telegram API."""
|
||||
|
||||
config = MagicMock()
|
||||
config.token = "test-token"
|
||||
bus = MagicMock()
|
||||
|
||||
channel = TelegramChannel(config, bus)
|
||||
|
||||
# Mock the internal _app and bot directly (skip start())
|
||||
mock_app = MagicMock()
|
||||
mock_bot = AsyncMock()
|
||||
mock_app.bot = mock_bot
|
||||
channel._app = mock_app
|
||||
|
||||
# Send a normal message
|
||||
msg = OutboundMessage(
|
||||
channel="telegram",
|
||||
chat_id="12345",
|
||||
content="Normal message",
|
||||
metadata={}
|
||||
)
|
||||
|
||||
await channel.send(msg)
|
||||
|
||||
# Verify bot.send_message WAS called
|
||||
mock_bot.send_message.assert_called_once()
|
||||
@@ -0,0 +1,505 @@
|
||||
# tests/test_visibility_signing.py
|
||||
import pytest
|
||||
from nanobot.agent.visibility import sign_content, verify_signature, has_forged_marker, strip_all_hidden_markers
|
||||
|
||||
def test_sign_content_adds_hmac_marker():
|
||||
"""Test that sign_content adds HMAC signature prefix."""
|
||||
content = "HEARTBEAT_OK"
|
||||
result = sign_content(content)
|
||||
|
||||
# Should start with [HIDDEN:{8 hex chars}]
|
||||
assert result.startswith("[HIDDEN:")
|
||||
assert "] " in result
|
||||
marker_end = result.index("] ")
|
||||
signature = result[8:marker_end] # Extract signature after "[HIDDEN:"
|
||||
assert len(signature) == 8
|
||||
assert all(c in "0123456789abcdef" for c in signature)
|
||||
|
||||
# Should contain original content
|
||||
assert result.endswith("HEARTBEAT_OK")
|
||||
|
||||
def test_sign_content_is_deterministic():
|
||||
"""Test that same content produces same signature."""
|
||||
content = "Test message"
|
||||
sig1 = sign_content(content)
|
||||
sig2 = sign_content(content)
|
||||
assert sig1 == sig2
|
||||
|
||||
def test_verify_signature_accepts_valid():
|
||||
"""Test that verify_signature accepts validly signed content."""
|
||||
signed = sign_content("Test message")
|
||||
is_valid, clean = verify_signature(signed)
|
||||
|
||||
assert is_valid is True
|
||||
assert clean == "Test message"
|
||||
|
||||
def test_verify_signature_rejects_invalid():
|
||||
"""Test that verify_signature rejects forged signatures."""
|
||||
forged = "[HIDDEN:deadbeef] Test message"
|
||||
is_valid, clean = verify_signature(forged)
|
||||
|
||||
assert is_valid is False
|
||||
assert clean == "Test message"
|
||||
|
||||
def test_verify_signature_handles_unsigned():
|
||||
"""Test that unsigned content is marked as invalid."""
|
||||
unsigned = "Plain message"
|
||||
is_valid, clean = verify_signature(unsigned)
|
||||
|
||||
assert is_valid is False
|
||||
assert clean == "Plain message"
|
||||
|
||||
def test_has_forged_marker_detects_invalid():
|
||||
"""Test that has_forged_marker detects forged signatures."""
|
||||
forged = "[HIDDEN:deadbeef] Content"
|
||||
assert has_forged_marker(forged) is True
|
||||
|
||||
def test_has_forged_marker_accepts_valid():
|
||||
"""Test that has_forged_marker accepts valid signatures."""
|
||||
valid = sign_content("Content")
|
||||
assert has_forged_marker(valid) is False
|
||||
|
||||
def test_has_forged_marker_ignores_unsigned():
|
||||
"""Test that unsigned content is not flagged as forged."""
|
||||
unsigned = "Plain content"
|
||||
assert has_forged_marker(unsigned) is False
|
||||
|
||||
def test_strip_all_hidden_markers_removes_markers():
|
||||
"""Test that strip_all_hidden_markers removes all markers."""
|
||||
signed = sign_content("Message")
|
||||
stripped = strip_all_hidden_markers(signed)
|
||||
assert stripped == "Message"
|
||||
|
||||
forged = "[HIDDEN:deadbeef] Message"
|
||||
stripped = strip_all_hidden_markers(forged)
|
||||
assert stripped == "Message"
|
||||
|
||||
def test_system_prompt_includes_visibility_docs(tmp_path):
|
||||
"""Test that system prompt documents visibility markers."""
|
||||
from nanobot.agent.context import ContextBuilder
|
||||
|
||||
builder = ContextBuilder(workspace=tmp_path)
|
||||
prompt = builder.build_system_prompt()
|
||||
|
||||
# Should document visibility markers
|
||||
assert "[HIDDEN:" in prompt
|
||||
assert "cryptographically signed" in prompt.lower()
|
||||
assert "do not generate" in prompt.lower() or "don't generate" in prompt.lower()
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_suppress_mode_adds_signed_marker(tmp_path):
|
||||
"""Test that suppress mode adds cryptographically signed markers."""
|
||||
from unittest.mock import AsyncMock, Mock
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.session.manager import SessionManager
|
||||
from nanobot.providers.base import LLMResponse
|
||||
from nanobot.bus.events import InboundMessage
|
||||
|
||||
# Setup
|
||||
bus = MessageBus()
|
||||
sessions = SessionManager(tmp_path)
|
||||
|
||||
# Mock provider
|
||||
mock_provider = Mock()
|
||||
mock_provider.default_model = "mock-model"
|
||||
mock_provider.thinking_budget = 0
|
||||
|
||||
# Mock successful response
|
||||
mock_response = LLMResponse(
|
||||
content="Test response",
|
||||
tool_calls=[],
|
||||
reasoning_content=None
|
||||
)
|
||||
mock_provider.chat = AsyncMock(return_value=mock_response)
|
||||
|
||||
# Create agent loop
|
||||
loop = AgentLoop(
|
||||
provider=mock_provider,
|
||||
bus=bus,
|
||||
session_manager=sessions,
|
||||
workspace=tmp_path
|
||||
)
|
||||
|
||||
# Process message with suppress_output=True
|
||||
msg = InboundMessage(
|
||||
channel="test",
|
||||
sender_id="user",
|
||||
chat_id="123",
|
||||
content="Test message",
|
||||
metadata={"suppress_output": True}
|
||||
)
|
||||
|
||||
response = await loop._process_message(msg)
|
||||
|
||||
# Verify response has suppressed metadata
|
||||
assert response.metadata.get("suppressed") is True
|
||||
|
||||
# Verify session contains signed marker
|
||||
session = sessions.get_or_create("test:123")
|
||||
assistant_messages = [m for m in session.messages if m.get("role") == "assistant"]
|
||||
assert len(assistant_messages) > 0
|
||||
|
||||
last_msg = assistant_messages[-1]["content"]
|
||||
assert last_msg.startswith("[HIDDEN:")
|
||||
|
||||
# Verify signature is valid
|
||||
is_valid, clean = verify_signature(last_msg)
|
||||
assert is_valid is True
|
||||
assert clean == "Test response"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_no_marker_accumulation_with_real_provider(tmp_path):
|
||||
"""
|
||||
CRITICAL TEST: Verify markers don't accumulate when model sees them in context.
|
||||
|
||||
This test uses a semi-realistic provider that sees the context and could
|
||||
potentially copy markers, unlike pure mocks that don't see context at all.
|
||||
"""
|
||||
from unittest.mock import AsyncMock, Mock
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.session.manager import SessionManager
|
||||
from nanobot.providers.base import LLMResponse
|
||||
from nanobot.bus.events import InboundMessage
|
||||
import re
|
||||
|
||||
# Setup
|
||||
bus = MessageBus()
|
||||
sessions = SessionManager(tmp_path)
|
||||
|
||||
# Create a provider that SEES context and simulates potential copying behavior
|
||||
class ContextAwareProvider:
|
||||
"""Provider that sees context and could copy markers (simulating real LLM)."""
|
||||
default_model = "test-model"
|
||||
thinking_budget = 0
|
||||
|
||||
def __init__(self):
|
||||
self.call_count = 0
|
||||
self.last_context = None
|
||||
|
||||
def get_default_model(self) -> str:
|
||||
"""Get the default model."""
|
||||
return self.default_model
|
||||
|
||||
async def chat(self, messages, **kwargs):
|
||||
self.call_count += 1
|
||||
self.last_context = messages
|
||||
|
||||
# Count markers in non-system messages (system prompt has 2 mentions in docs)
|
||||
marker_count = sum(
|
||||
msg.get("content", "").count("[HIDDEN:")
|
||||
for msg in messages
|
||||
if isinstance(msg.get("content"), str) and msg.get("role") != "system"
|
||||
)
|
||||
|
||||
# Simulate model behavior: on first call (msg 2), sees 1 marker from msg 1
|
||||
# The model should NOT copy it
|
||||
if self.call_count == 2:
|
||||
# Verify context has exactly 1 marker in assistant messages (from message 1)
|
||||
assert marker_count == 1, f"Expected 1 marker in context, found {marker_count}"
|
||||
|
||||
# Always return clean response (good model behavior)
|
||||
return LLMResponse(
|
||||
content=f"Response {self.call_count}",
|
||||
tool_calls=[],
|
||||
reasoning_content=None
|
||||
)
|
||||
|
||||
provider = ContextAwareProvider()
|
||||
|
||||
# Create agent loop
|
||||
loop = AgentLoop(
|
||||
provider=provider,
|
||||
bus=bus,
|
||||
session_manager=sessions,
|
||||
workspace=tmp_path
|
||||
)
|
||||
|
||||
# Use unique chat_id for this test to avoid pollution from previous runs
|
||||
import time
|
||||
test_chat_id = f"test_accumulation_{int(time.time()*1000)}"
|
||||
|
||||
# Message 1: suppress_output=True → should add signed marker
|
||||
msg1 = InboundMessage(
|
||||
channel="test",
|
||||
sender_id="user",
|
||||
chat_id=test_chat_id,
|
||||
content="Hidden message 1",
|
||||
metadata={"suppress_output": True}
|
||||
)
|
||||
|
||||
await loop._process_message(msg1)
|
||||
|
||||
# Verify message 1 has signed marker
|
||||
session = sessions.get_or_create(f"test:{test_chat_id}")
|
||||
assistant_msgs = [m for m in session.messages if m.get("role") == "assistant"]
|
||||
assert len(assistant_msgs) == 1
|
||||
msg1_content = assistant_msgs[0]["content"]
|
||||
assert msg1_content.startswith("[HIDDEN:")
|
||||
is_valid, clean = verify_signature(msg1_content)
|
||||
assert is_valid is True
|
||||
assert clean == "Response 1"
|
||||
|
||||
# Count markers in session after message 1
|
||||
marker_count_1 = sum(m.get("content", "").count("[HIDDEN:") for m in session.messages if isinstance(m.get("content"), str))
|
||||
assert marker_count_1 == 1, f"Expected 1 marker after msg1, found {marker_count_1}"
|
||||
|
||||
# Message 2: Normal message (context includes message 1 with marker)
|
||||
msg2 = InboundMessage(
|
||||
channel="test",
|
||||
sender_id="user",
|
||||
chat_id=test_chat_id,
|
||||
content="Normal message 2",
|
||||
metadata={}
|
||||
)
|
||||
|
||||
await loop._process_message(msg2)
|
||||
|
||||
# Verify message 2 response does NOT start with [HIDDEN: (model didn't copy)
|
||||
session = sessions.get_or_create(f"test:{test_chat_id}")
|
||||
assistant_msgs = [m for m in session.messages if m.get("role") == "assistant"]
|
||||
assert len(assistant_msgs) == 2
|
||||
msg2_content = assistant_msgs[1]["content"]
|
||||
assert not msg2_content.startswith("[HIDDEN:"), f"Message 2 should not start with [HIDDEN:, got: {msg2_content}"
|
||||
|
||||
# Verify still only 1 marker in session (no accumulation)
|
||||
marker_count_2 = sum(m.get("content", "").count("[HIDDEN:") for m in session.messages if isinstance(m.get("content"), str))
|
||||
assert marker_count_2 == 1, f"Expected 1 marker after msg2, found {marker_count_2} (ACCUMULATION DETECTED)"
|
||||
|
||||
# Message 3: Another suppress_output=True → should add SECOND signed marker
|
||||
msg3 = InboundMessage(
|
||||
channel="test",
|
||||
sender_id="user",
|
||||
chat_id=test_chat_id,
|
||||
content="Hidden message 3",
|
||||
metadata={"suppress_output": True}
|
||||
)
|
||||
|
||||
await loop._process_message(msg3)
|
||||
|
||||
# Verify message 3 has signed marker
|
||||
session = sessions.get_or_create(f"test:{test_chat_id}")
|
||||
assistant_msgs = [m for m in session.messages if m.get("role") == "assistant"]
|
||||
assert len(assistant_msgs) == 3
|
||||
msg3_content = assistant_msgs[2]["content"]
|
||||
assert msg3_content.startswith("[HIDDEN:")
|
||||
is_valid, clean = verify_signature(msg3_content)
|
||||
assert is_valid is True
|
||||
assert clean == "Response 3"
|
||||
|
||||
# Verify exactly 2 markers in session (one from msg1, one from msg3)
|
||||
marker_count_3 = sum(m.get("content", "").count("[HIDDEN:") for m in session.messages if isinstance(m.get("content"), str))
|
||||
assert marker_count_3 == 2, f"Expected 2 markers after msg3, found {marker_count_3}"
|
||||
|
||||
# CRITICAL: Verify no double/triple markers like "[HIDDEN: [HIDDEN: [HIDDEN: message"
|
||||
for msg in session.messages:
|
||||
content = msg.get("content", "")
|
||||
if isinstance(content, str) and "[HIDDEN:" in content:
|
||||
# Count occurrences of [HIDDEN: pattern in this single message
|
||||
hidden_count = content.count("[HIDDEN:")
|
||||
assert hidden_count == 1, f"Message has {hidden_count} [HIDDEN: markers (accumulation): {content[:100]}"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_forged_marker_triggers_rejection(tmp_path):
|
||||
"""Test that forged markers trigger rejection and retry."""
|
||||
from unittest.mock import AsyncMock, Mock
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.session.manager import SessionManager
|
||||
from nanobot.providers.base import LLMResponse
|
||||
from nanobot.bus.events import InboundMessage
|
||||
|
||||
# Setup
|
||||
bus = MessageBus()
|
||||
sessions = SessionManager(tmp_path)
|
||||
|
||||
# Mock provider
|
||||
mock_provider = Mock()
|
||||
mock_provider.default_model = "mock-model"
|
||||
mock_provider.thinking_budget = 0
|
||||
|
||||
# First response: model tries to forge marker
|
||||
forged_response = LLMResponse(
|
||||
content="[HIDDEN:deadbeef] Forged message",
|
||||
tool_calls=[],
|
||||
reasoning_content=None
|
||||
)
|
||||
|
||||
# Second response: clean response after correction
|
||||
clean_response = LLMResponse(
|
||||
content="Clean message",
|
||||
tool_calls=[],
|
||||
reasoning_content=None
|
||||
)
|
||||
|
||||
mock_provider.chat = AsyncMock(side_effect=[forged_response, clean_response])
|
||||
|
||||
# Create agent loop
|
||||
loop = AgentLoop(
|
||||
provider=mock_provider,
|
||||
bus=bus,
|
||||
session_manager=sessions,
|
||||
workspace=tmp_path
|
||||
)
|
||||
|
||||
# Process message with suppress_output=True
|
||||
msg = InboundMessage(
|
||||
channel="test",
|
||||
sender_id="user",
|
||||
chat_id="123",
|
||||
content="Test message",
|
||||
metadata={"suppress_output": True}
|
||||
)
|
||||
|
||||
response = await loop._process_message(msg)
|
||||
|
||||
# Verify provider.chat was called twice (initial + retry)
|
||||
assert mock_provider.chat.call_count == 2
|
||||
|
||||
# Verify second call included correction message
|
||||
second_call_messages = mock_provider.chat.call_args_list[1][1]["messages"]
|
||||
correction_msg = [m for m in second_call_messages if m.get("role") == "user" and "rejected" in m.get("content", "").lower()]
|
||||
assert len(correction_msg) > 0
|
||||
|
||||
# Verify final response uses clean content (not forged)
|
||||
session = sessions.get_or_create("test:123")
|
||||
assistant_messages = [m for m in session.messages if m.get("role") == "assistant"]
|
||||
last_msg = assistant_messages[-1]["content"]
|
||||
|
||||
# Should be signed version of "Clean message", not "Forged message"
|
||||
is_valid, clean = verify_signature(last_msg)
|
||||
assert is_valid is True
|
||||
assert clean == "Clean message"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_system_message_handler_uses_signed_markers(tmp_path):
|
||||
"""Test that _process_system_message uses signed markers in suppress mode."""
|
||||
from unittest.mock import AsyncMock, Mock
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.session.manager import SessionManager
|
||||
from nanobot.providers.base import LLMResponse
|
||||
from nanobot.bus.events import InboundMessage
|
||||
|
||||
# Setup
|
||||
bus = MessageBus()
|
||||
sessions = SessionManager(tmp_path)
|
||||
|
||||
# Mock provider
|
||||
mock_provider = Mock()
|
||||
mock_provider.default_model = "mock-model"
|
||||
mock_provider.thinking_budget = 0
|
||||
mock_response = LLMResponse(
|
||||
content="System response",
|
||||
tool_calls=[],
|
||||
reasoning_content=None
|
||||
)
|
||||
mock_provider.chat = AsyncMock(return_value=mock_response)
|
||||
|
||||
# Create agent loop
|
||||
loop = AgentLoop(
|
||||
provider=mock_provider,
|
||||
bus=bus,
|
||||
session_manager=sessions,
|
||||
workspace=tmp_path
|
||||
)
|
||||
|
||||
# Process system message with suppress_output=True
|
||||
msg = InboundMessage(
|
||||
channel="system",
|
||||
sender_id="subagent",
|
||||
chat_id="test:123",
|
||||
content="[Subagent completed] Result: OK",
|
||||
metadata={"suppress_output": True}
|
||||
)
|
||||
|
||||
response = await loop._process_system_message(msg)
|
||||
|
||||
# Verify response has suppressed metadata
|
||||
assert response.metadata.get("suppressed") is True
|
||||
|
||||
# Verify session contains signed marker
|
||||
session = sessions.get_or_create("test:123")
|
||||
assistant_messages = [m for m in session.messages if m.get("role") == "assistant"]
|
||||
assert len(assistant_messages) > 0
|
||||
|
||||
last_msg = assistant_messages[-1]["content"]
|
||||
assert last_msg.startswith("[HIDDEN:")
|
||||
|
||||
# Verify signature is valid
|
||||
is_valid, clean = verify_signature(last_msg)
|
||||
assert is_valid is True
|
||||
assert clean == "System response"
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_system_message_handler_rejects_forged_markers(tmp_path):
|
||||
"""Test that _process_system_message rejects forged markers."""
|
||||
from unittest.mock import AsyncMock, Mock
|
||||
from nanobot.agent.loop import AgentLoop
|
||||
from nanobot.bus.queue import MessageBus
|
||||
from nanobot.session.manager import SessionManager
|
||||
from nanobot.providers.base import LLMResponse
|
||||
from nanobot.bus.events import InboundMessage
|
||||
|
||||
# Setup
|
||||
bus = MessageBus()
|
||||
sessions = SessionManager(tmp_path)
|
||||
|
||||
# Mock provider
|
||||
mock_provider = Mock()
|
||||
mock_provider.default_model = "mock-model"
|
||||
mock_provider.thinking_budget = 0
|
||||
|
||||
# First response: model tries to forge marker
|
||||
forged_response = LLMResponse(
|
||||
content="[HIDDEN:deadbeef] Forged system message",
|
||||
tool_calls=[],
|
||||
reasoning_content=None
|
||||
)
|
||||
|
||||
# Second response: clean response after correction
|
||||
clean_response = LLMResponse(
|
||||
content="Clean system response",
|
||||
tool_calls=[],
|
||||
reasoning_content=None
|
||||
)
|
||||
|
||||
mock_provider.chat = AsyncMock(side_effect=[forged_response, clean_response])
|
||||
|
||||
# Create agent loop
|
||||
loop = AgentLoop(
|
||||
provider=mock_provider,
|
||||
bus=bus,
|
||||
session_manager=sessions,
|
||||
workspace=tmp_path
|
||||
)
|
||||
|
||||
# Process system message with suppress_output=True
|
||||
msg = InboundMessage(
|
||||
channel="system",
|
||||
sender_id="subagent",
|
||||
chat_id="test:456",
|
||||
content="[Subagent completed] Result: OK",
|
||||
metadata={"suppress_output": True}
|
||||
)
|
||||
|
||||
response = await loop._process_system_message(msg)
|
||||
|
||||
# Verify provider.chat was called twice (initial + retry)
|
||||
assert mock_provider.chat.call_count == 2
|
||||
|
||||
# Verify second call included correction message
|
||||
second_call_messages = mock_provider.chat.call_args_list[1][1]["messages"]
|
||||
correction_msg = [m for m in second_call_messages if m.get("role") == "user" and "rejected" in m.get("content", "").lower()]
|
||||
assert len(correction_msg) > 0
|
||||
|
||||
# Verify final response uses clean content (not forged)
|
||||
session = sessions.get_or_create("test:456")
|
||||
assistant_messages = [m for m in session.messages if m.get("role") == "assistant"]
|
||||
last_msg = assistant_messages[-1]["content"]
|
||||
|
||||
# Should be signed version of "Clean system response", not "Forged system message"
|
||||
is_valid, clean = verify_signature(last_msg)
|
||||
assert is_valid is True
|
||||
assert clean == "Clean system response"
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user