"""Tools a playbook can call during chat. Ollama drives the calling: `/api/chat` with a `tools` param returns `message.tool_calls`, and this module is just the registry + dispatch. Most tools READ local state (memory, history, documents, models). Some act: `web_search`/`fetch_url` make outbound HTTP requests, `remember` writes a memory fact, and `curry_*` reads or writes NexusOS's vendored Curry ledger. The per-playbook allowlist (`PlaybookItem.tools`) is the first gate. Curry write/execute tools additionally require per-call approval when model-issued. A message consisting only of `/tool_name(arg=val, ...)` dispatches directly; see `synapse/slash_commands.py` for that explicit-human-command boundary. """ from __future__ import annotations import json from typing import Awaitable, Callable from .curry_store import curry_db from .memory.store import store, MemoryItem from .ollama_manager import get_ollama_manager async def _search_memory(query: str = "", **_) -> str: q = (query or "").strip().lower() hits = [ {"section": it.section, "text": it.text} for it in store.all() if not q or q in it.text.lower() or q in (it.section or "").lower() or any(q in t.lower() for t in it.tags) ] return json.dumps(hits[:20]) async def _search_history(query: str = "", **_) -> str: # Hybrid recall: semantic (embeddings) unioned with lexical, falls back to # lexical if embeddings are down. Same retrieval the chat endpoint uses. convs = await store.semantic_search_conversations( query or "", get_ollama_manager().embed, limit=3 ) return json.dumps([{"matches": c.get("matches", [])} for c in convs]) async def _list_models(**_) -> str: return json.dumps(await get_ollama_manager().list_models()) async def _search_documents(query: str = "", **_) -> str: hits = await store.search_documents(query or "", get_ollama_manager().embed, limit=3) return json.dumps([{"title": h["title"], "text": h["text"]} for h in hits]) async def _get_time(**_) -> str: from datetime import datetime return json.dumps({"now": datetime.now().isoformat(timespec="seconds")}) async def _web_search(query: str = "", **_) -> str: import asyncio as _a from .search import web_search res = await _a.to_thread(web_search, query or "", 4) return res or "(no results)" _FETCH_MAX_REDIRECTS = 5 def _ip_is_blocked(ip: str) -> bool: """True if an address is one an outbound fetch has no business reaching: loopback, RFC1918/ULA private, link-local (incl. 169.254.169.254 cloud metadata), multicast, reserved, or unspecified. IPv4-mapped IPv6 is unwrapped first so ::ffff:127.0.0.1 can't sneak a loopback past the check.""" import ipaddress try: addr = ipaddress.ip_address(ip.split("%")[0]) # drop any IPv6 zone id except ValueError: return True # unparseable -> refuse rather than guess mapped = getattr(addr, "ipv4_mapped", None) if mapped is not None: addr = mapped return ( addr.is_loopback or addr.is_private or addr.is_link_local or addr.is_multicast or addr.is_reserved or addr.is_unspecified ) def _ssrf_guard(host: str) -> str | None: """Resolve a hostname and return an error string if ANY of its A/AAAA records is a blocked address, else None. Checking every answer stops a name from smuggling one private record alongside a public one. ponytail: this validates then httpx re-resolves on connect, so a sub-second DNS-rebind could still slip a private address through the TOCTOU gap. That's an advanced attack against a playbook-gated, single-user tool; pin the connection to the resolved IP if this ever faces untrusted callers.""" import socket if not host: return "missing host" try: infos = socket.getaddrinfo(host, None) except socket.gaierror as e: return f"cannot resolve host: {e}" ips = {info[4][0] for info in infos} if not ips: return "host did not resolve" blocked = [ip for ip in ips if _ip_is_blocked(ip)] if blocked: return f"refusing to fetch a private/loopback/link-local address ({', '.join(sorted(blocked))})" return None async def _fetch_url(url: str = "", **_) -> str: import re import httpx from urllib.parse import urlparse, urljoin url = (url or "").strip() if not url.startswith(("http://", "https://")): return json.dumps({"error": "url must start with http:// or https://"}) # SSRF guard: validate the host of the initial URL AND every redirect hop # against the private/loopback/link-local block-list before connecting, so a # granted fetch_url can't be steered at 127.0.0.1:11434, cloud metadata, or # LAN hosts — and a public URL can't 302 its way there either. try: async with httpx.AsyncClient(timeout=15.0, follow_redirects=False) as c: for _ in range(_FETCH_MAX_REDIRECTS + 1): parsed = urlparse(url) if parsed.scheme not in ("http", "https"): return json.dumps({"error": "only http(s) URLs are allowed"}) err = _ssrf_guard(parsed.hostname or "") if err: return json.dumps({"error": f"blocked: {err}"}) r = await c.get(url, headers={"User-Agent": "NexusOS/1.0"}) location = r.headers.get("location") if r.is_redirect and location: url = urljoin(url, location) continue r.raise_for_status() html = r.text break else: return json.dumps({"error": "too many redirects"}) except Exception as e: return json.dumps({"error": f"fetch failed: {e}"}) text = re.sub(r"(?is)<(script|style).*?", " ", html) text = re.sub(r"(?s)<[^>]+>", " ", text) text = re.sub(r"\s+", " ", text).strip() return text[:4000] async def _remember(text: str = "", section: str = "General", **_) -> str: """WRITE tool: persist a memory fact. First action tool — allowlist-gated.""" import uuid as _uuid text = (text or "").strip() if not text: return json.dumps({"error": "text is required"}) store.add(MemoryItem(id=str(_uuid.uuid4()), section=(section or "General"), text=text)) return json.dumps({"saved": text, "section": section or "General"}) # --- Repo file access (read-only, scoped to PROJECT_ROOT) ------------------- # Paths never leave the repo: every request is resolve()d and checked against # PROJECT_ROOT, which also kills symlink escapes. _DENIED covers the parts of # the tree that are either secrets, private data, or multi-GB noise. _DENIED = { ".git", ".env", "Promethean", "node_modules", "models", "ollama", "runtime", "dist", "__pycache__", ".git-credentials", } _READ_MAX = 60_000 def _repo_path(rel: str) -> "tuple[object, str | None]": """Resolve a repo-relative path. Returns (path, error-string).""" from .nexus_config import PROJECT_ROOT rel = (rel or "").strip().lstrip("/") if not rel: return None, "path is required" target = (PROJECT_ROOT / rel).resolve() if not target.is_relative_to(PROJECT_ROOT): return None, "path escapes the project root" parts = set(target.relative_to(PROJECT_ROOT).parts) if parts & _DENIED or target.name.endswith((".db", ".db.sql", ".pem", ".key")): return None, f"{rel} is not readable" return target, None async def _read_file(path: str = "", **_) -> str: target, err = _repo_path(path) if err: return json.dumps({"error": err}) if not target.is_file(): return json.dumps({"error": f"{path} does not exist"}) try: text = target.read_text(encoding="utf-8", errors="replace") except OSError as e: return json.dumps({"error": f"cannot read {path}: {e}"}) return json.dumps({ "path": path, "truncated": len(text) > _READ_MAX, "content": text[:_READ_MAX], }) async def _list_files(pattern: str = "", **_) -> str: """Glob the repo so the model discovers real paths instead of inventing them.""" from .nexus_config import PROJECT_ROOT pattern = (pattern or "**/*.py").strip().lstrip("/") hits = [] for f in PROJECT_ROOT.glob(pattern): if not f.is_file(): continue target, err = _repo_path(str(f.relative_to(PROJECT_ROOT))) if err: continue hits.append(str(f.relative_to(PROJECT_ROOT))) if len(hits) >= 200: break return json.dumps(sorted(hits)) # Curry (synapse/curry_core.py, vendored) — immutable, versioned constants and # functions. Expected caller errors keep the same structured JSON shape as the # other tools instead of falling through dispatch()'s generic error envelope. _CURRY_FENCE_LANG = "nexus-curry" def _curry_fence(payload: dict) -> str: body = json.dumps(payload, ensure_ascii=False, default=str).replace("`", "\\u0060") return f"```{_CURRY_FENCE_LANG}\n{body}\n```" async def _curry_call(fn, *args, **kwargs) -> dict: # Curry holds one SQLite connection. Calls stay on the event-loop thread, # where these local database operations are short and naturally serialized. try: result = fn(*args, **kwargs) return {"ok": True, "result": result} except (KeyError, ValueError, TypeError, RuntimeError) as exc: return {"ok": False, "error": str(exc)} async def _curry_declare_constant( id: str = "", version: int = 0, value=None, type_signature: str = "", description: str = "", **_, ) -> str: """ACTION tool: declare a new, immutable version of a named constant.""" out = await _curry_call( curry_db.declare_constant, id, version, value, type_signature, description or None ) if out["ok"]: out = {"ok": True, "id": id, "version": version} out["fence"] = _curry_fence({"kind": "declare_constant", **out}) return json.dumps(out) async def _curry_get_constant(id: str = "", version: int = 0, **_) -> str: return json.dumps(await _curry_call(curry_db.get_constant, id, version)) async def _curry_get_constant_latest(id: str = "", **_) -> str: return json.dumps(await _curry_call(curry_db.get_constant_latest, id)) async def _curry_list_constants(active_only: bool = True, **_) -> str: return json.dumps(await _curry_call(curry_db.list_constants, active_only)) async def _curry_retire_constant( id: str = "", version: int = 0, reason: str = "", **_, ) -> str: out = await _curry_call( curry_db.retire_constant_with_reason, id, version, reason or "retired via tool call", ) return json.dumps(out) async def _curry_declare_function( name: str = "", version: int = 0, body: str = "", constant_bindings: dict | None = None, function_bindings: dict | None = None, is_pure: bool = False, expected_args: list | None = None, description: str = "", arg_descriptions: dict | None = None, **_, ) -> str: """ACTION tool: declare one statically validated expression.""" out = await _curry_call( curry_db.declare_function, name, version, body, constant_bindings or {}, function_bindings or {}, is_pure, expected_args, description or None, arg_descriptions, ) if out["ok"]: out = {"ok": True, "name": name, "version": version} out["fence"] = _curry_fence({"kind": "declare_function", **out}) return json.dumps(out) async def _curry_get_function(name: str = "", version: int = 0, **_) -> str: return json.dumps(await _curry_call(curry_db.get_function, name, version)) async def _curry_list_functions(active_only: bool = True, **_) -> str: return json.dumps(await _curry_call(curry_db.list_functions, active_only)) async def _curry_call_function( name: str = "", version: int = 0, args: dict | None = None, **_, ) -> str: out = await _curry_call(curry_db.call_function, name, version, args or {}) if out["ok"]: out["fence"] = _curry_fence({ "kind": "call_function", "name": name, "version": version, **out, }) return json.dumps(out) async def _curry_retire_function( name: str = "", version: int = 0, reason: str = "", **_, ) -> str: out = await _curry_call( curry_db.retire_function_with_reason, name, version, reason or "retired via tool call", ) return json.dumps(out) # name -> (schema, callable). Schema is the OpenAI/Ollama function-tool format. REGISTRY: dict[str, tuple[dict, Callable[..., Awaitable[str]]]] = { "search_memory": ( { "type": "function", "function": { "name": "search_memory", "description": "Search the user's persistent memory facts. Empty query returns all facts.", "parameters": { "type": "object", "properties": {"query": {"type": "string", "description": "text to match"}}, }, }, }, _search_memory, ), "search_history": ( { "type": "function", "function": { "name": "search_history", "description": "Search past conversations for exchanges containing the query text.", "parameters": { "type": "object", "properties": {"query": {"type": "string"}}, "required": ["query"], }, }, }, _search_history, ), "list_models": ( { "type": "function", "function": { "name": "list_models", "description": "List the locally installed Ollama models.", "parameters": {"type": "object", "properties": {}}, }, }, _list_models, ), "read_file": ( { "type": "function", "function": { "name": "read_file", "description": "Read a source file from the NexusOS repository. Path is relative to the project root, e.g. 'synapse/main.py'.", "parameters": { "type": "object", "properties": {"path": {"type": "string", "description": "repo-relative file path"}}, "required": ["path"], }, }, }, _read_file, ), "list_files": ( { "type": "function", "function": { "name": "list_files", "description": "List files in the NexusOS repository matching a glob, e.g. 'synapse/**/*.py' or 'interface/web/src/*.jsx'. Use this to find real paths before reading.", "parameters": { "type": "object", "properties": {"pattern": {"type": "string", "description": "glob relative to the project root"}}, }, }, }, _list_files, ), "search_documents": ( { "type": "function", "function": { "name": "search_documents", "description": "Search the user's uploaded documents for relevant passages.", "parameters": { "type": "object", "properties": {"query": {"type": "string"}}, "required": ["query"], }, }, }, _search_documents, ), "get_time": ( { "type": "function", "function": { "name": "get_time", "description": "Get the current local date and time.", "parameters": {"type": "object", "properties": {}}, }, }, _get_time, ), "web_search": ( { "type": "function", "function": { "name": "web_search", "description": "Search the web (DuckDuckGo) and return the top result snippets.", "parameters": { "type": "object", "properties": {"query": {"type": "string"}}, "required": ["query"], }, }, }, _web_search, ), "fetch_url": ( { "type": "function", "function": { "name": "fetch_url", "description": "Fetch a web page and return its visible text (truncated).", "parameters": { "type": "object", "properties": {"url": {"type": "string", "description": "http(s) URL"}}, "required": ["url"], }, }, }, _fetch_url, ), "remember": ( { "type": "function", "function": { "name": "remember", "description": "Save a durable fact to the user's persistent memory.", "parameters": { "type": "object", "properties": { "text": {"type": "string", "description": "the fact to remember"}, "section": {"type": "string", "description": "optional category, e.g. Health"}, }, "required": ["text"], }, }, }, _remember, ), "curry_declare_constant": ( { "type": "function", "function": { "name": "curry_declare_constant", "description": ( "Declare a new immutable version of a Curry constant. " "Requires per-call human approval when model-issued." ), "parameters": { "type": "object", "properties": { "id": {"type": "string", "description": "Constant identifier."}, "version": {"type": "integer", "description": "A new, higher version."}, "value": {"description": "Value matching type_signature."}, "type_signature": { "type": "string", "description": ( "Float64 | Int32 | String | Blob | Json | Tokens | " "Currency | Bool" ), }, "description": {"type": "string"}, }, "required": ["id", "version", "value", "type_signature"], }, }, }, _curry_declare_constant, ), "curry_get_constant": ( { "type": "function", "function": { "name": "curry_get_constant", "description": "Retrieve a Curry constant by exact id and version.", "parameters": { "type": "object", "properties": { "id": {"type": "string"}, "version": {"type": "integer"}, }, "required": ["id", "version"], }, }, }, _curry_get_constant, ), "curry_get_constant_latest": ( { "type": "function", "function": { "name": "curry_get_constant_latest", "description": "Retrieve the latest active version of a Curry constant.", "parameters": { "type": "object", "properties": {"id": {"type": "string"}}, "required": ["id"], }, }, }, _curry_get_constant_latest, ), "curry_list_constants": ( { "type": "function", "function": { "name": "curry_list_constants", "description": "List Curry constants.", "parameters": { "type": "object", "properties": {"active_only": {"type": "boolean"}}, }, }, }, _curry_list_constants, ), "curry_retire_constant": ( { "type": "function", "function": { "name": "curry_retire_constant", "description": ( "Retire, but do not delete, a Curry constant version. " "Requires per-call human approval when model-issued." ), "parameters": { "type": "object", "properties": { "id": {"type": "string"}, "version": {"type": "integer"}, "reason": {"type": "string"}, }, "required": ["id", "version"], }, }, }, _curry_retire_constant, ), "curry_declare_function": ( { "type": "function", "function": { "name": "curry_declare_function", "description": ( "Declare a new immutable Curry function version. The body is one " "statically validated Python expression. Requires per-call human " "approval when model-issued." ), "parameters": { "type": "object", "properties": { "name": {"type": "string"}, "version": {"type": "integer"}, "body": {"type": "string"}, "constant_bindings": {"type": "object"}, "function_bindings": {"type": "object"}, "is_pure": {"type": "boolean"}, "expected_args": { "type": "array", "items": {"type": "string"}, }, "description": {"type": "string"}, "arg_descriptions": {"type": "object"}, }, "required": ["name", "version", "body"], }, }, }, _curry_declare_function, ), "curry_get_function": ( { "type": "function", "function": { "name": "curry_get_function", "description": "Retrieve a Curry function by exact name and version.", "parameters": { "type": "object", "properties": { "name": {"type": "string"}, "version": {"type": "integer"}, }, "required": ["name", "version"], }, }, }, _curry_get_function, ), "curry_list_functions": ( { "type": "function", "function": { "name": "curry_list_functions", "description": "List Curry functions and their expected arguments.", "parameters": { "type": "object", "properties": {"active_only": {"type": "boolean"}}, }, }, }, _curry_list_functions, ), "curry_call_function": ( { "type": "function", "function": { "name": "curry_call_function", "description": ( "Execute an exact Curry function version with runtime arguments. " "Requires per-call human approval when model-issued." ), "parameters": { "type": "object", "properties": { "name": {"type": "string"}, "version": {"type": "integer"}, "args": {"type": "object"}, }, "required": ["name", "version"], }, }, }, _curry_call_function, ), "curry_retire_function": ( { "type": "function", "function": { "name": "curry_retire_function", "description": ( "Retire, but do not delete, a Curry function version. " "Requires per-call human approval when model-issued." ), "parameters": { "type": "object", "properties": { "name": {"type": "string"}, "version": {"type": "integer"}, "reason": {"type": "string"}, }, "required": ["name", "version"], }, }, }, _curry_retire_function, ), } # Tools that act (write local state or reach the network). These require an # explicit consent gate (settings.allow_action_tools) on top of the per-playbook # allowlist — a playbook granting one isn't enough on its own. Curry writes and # execution additionally require per-call approval for model-issued calls. CURRY_ALWAYS_ASK_TOOLS = frozenset({ "curry_declare_constant", "curry_retire_constant", "curry_declare_function", "curry_retire_function", "curry_call_function", }) ACTION_TOOLS = frozenset({"web_search", "fetch_url", "remember"}) | CURRY_ALWAYS_ASK_TOOLS ALWAYS_ASK_ACTION_TOOLS = CURRY_ALWAYS_ASK_TOOLS def is_action(name: str) -> bool: return name in ACTION_TOOLS def schemas_for(names: list[str], allow_actions: bool = True) -> list[dict]: """Tool schemas for a playbook's allowlist; unknown names are dropped. When allow_actions is False, action tools are withheld so the model can't even call them.""" return [ REGISTRY[n][0] for n in (names or []) if n in REGISTRY and (allow_actions or not is_action(n)) ] async def dispatch(name: str, args: dict | None) -> str: """Run a tool by name. Never raises — returns an error string on failure.""" entry = REGISTRY.get(name) if not entry: return json.dumps({"error": f"unknown tool: {name}"}) try: return await entry[1](**(args or {})) except Exception as e: # a broken tool must not kill the chat loop return json.dumps({"error": f"{name} failed: {e}"})