"""manuals/ must stay in step with the tools that exist — and with the docs that cite it.
A manual is filed under the tool's own name: ``compile_scad.md`` documents
``mcp__navi-3d__compile_scad``. That name is the only thing that lets tool_manual
find the file, and nothing enforced it. So three manuals documented tools that are
called something else — ``write_mcp_server.md``/``write_tool.md`` (no such tools at
all), ``model_3d.md`` (the tool is ``compile_scad``), ``render_3d.md`` (the tool is
``render_stl``) — and the hand-written text was dead weight while the agent got a
schema dump in its place. The prompts pointed at those same wrong names, so fixing
the files without fixing the callers would have moved the breakage, not removed it.
These tests read the real repository tree, not a fixture: they are the guard on the
files as shipped, and they are what fails when the next tool is renamed.
"""
from __future__ import annotations
import ast
import difflib
import json
import re
from pathlib import Path
import pytest
from navi.core.registry import ToolRegistry
from navi.tools import tool_manual
from navi.tools.tool_manual import ToolManualTool
from tests.conftest_factory import FakeTool
REPO_ROOT = Path(tool_manual.__file__).resolve().parents[2]
# ── where a real tool name can come from ─────────────────────────────────
def _string_name_assignments(tree: ast.AST) -> set[str]:
"""Every ``name = "literal"`` at module or class level — how tools declare themselves."""
found: set[str] = set()
for node in ast.walk(tree):
if not isinstance(node, (ast.Module, ast.ClassDef)):
continue
for stmt in node.body:
if isinstance(stmt, ast.Assign) and isinstance(stmt.value, ast.Constant):
value = stmt.value.value
targets = [t for t in stmt.targets if isinstance(t, ast.Name)]
elif isinstance(stmt, ast.AnnAssign) and isinstance(stmt.value, ast.Constant):
value = stmt.value.value
targets = [stmt.target] if isinstance(stmt.target, ast.Name) else []
else:
continue
if isinstance(value, str) and value and any(t.id == "name" for t in targets):
found.add(value)
return found
def _builtin_tool_names() -> set[str]:
"""Built-ins (navi/tools/) and user tools (tools/) — _-prefixed scaffolds are skipped."""
names: set[str] = set()
for pattern in ("navi/tools/*.py", "navi/tools/*/*.py", "tools/*.py"):
for path in REPO_ROOT.glob(pattern):
if path.name.startswith("_"):
continue
try:
names |= _string_name_assignments(ast.parse(path.read_text(encoding="utf-8")))
except SyntaxError: # a file the interpreter would reject is not a tool
continue
return names
def _declared_mcp_tool_names() -> set[str]:
"""``@mcp.tool(name="x")`` in the MCP server sources — the tools a server ships."""
names: set[str] = set()
for path in REPO_ROOT.glob("mcp-servers/*/app/mcp_server.py"):
tree = ast.parse(path.read_text(encoding="utf-8"))
for node in ast.walk(tree):
if not isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
continue
for decorator in node.decorator_list:
if not isinstance(decorator, ast.Call):
continue
func = decorator.func
if not (isinstance(func, ast.Attribute) and func.attr == "tool"):
continue
for kw in decorator.keywords:
if (
kw.arg == "name"
and isinstance(kw.value, ast.Constant)
and isinstance(kw.value.value, str)
):
names.add(kw.value.value)
return names
def _configured_mcp_tools() -> dict[str, set[str]]:
"""server name -> tool names, from the tracked server configs.
The server name is the config's filename (``navi/mcp/config.py`` takes it from
``file_path.stem``), which is what makes ``mcp__<server>__<tool>`` computable here.
"""
servers: dict[str, set[str]] = {}
for path in sorted((REPO_ROOT / "mcp_servers.d").glob("*.json")):
groups = json.loads(path.read_text(encoding="utf-8")).get("groups") or {}
names = {
name
for group in groups.values()
if isinstance(group, list)
for name in group
if isinstance(name, str)
}
if names:
servers[path.stem] = names
return servers
NATIVE_TOOLS = _builtin_tool_names()
MCP_BY_SERVER = _configured_mcp_tools()
# The names the agent calls by (mcp__<server>__<tool>) and the bare ones it may ask
# about instead. Both are real spellings, so both count as "this tool exists".
MCP_FULL_NAMES = {f"mcp__{server}__{tool}" for server, tools in MCP_BY_SERVER.items() for tool in tools}
MCP_BARE_NAMES = (
{tool for tools in MCP_BY_SERVER.values() for tool in tools} | _declared_mcp_tool_names()
)
KNOWN_TOOLS = NATIVE_TOOLS | MCP_BARE_NAMES | MCP_FULL_NAMES
def _full_names_for(tool: str) -> list[str]:
"""Every mcp__<server>__<tool> a bare name may legitimately mean (usually one)."""
return sorted(
f"mcp__{server}__{tool}" for server, tools in MCP_BY_SERVER.items() if tool in tools
)
def _manuals() -> list[Path]:
return sorted(tool_manual.MANUALS_DIR.glob("*.md"))
def _manual_names_on_disk() -> set[str]:
return {path.stem for path in _manuals()}
def _closest(name: str) -> str:
matches = difflib.get_close_matches(name, sorted(KNOWN_TOOLS), n=3, cutoff=0.5)
return f" Closest real tool names: {', '.join(matches)}." if matches else ""
@pytest.fixture(scope="module")
def registry() -> ToolRegistry:
"""Every tool that exists, registered — so the index can bucket by real source.
Only the names the registry really holds: a bare MCP name is a spelling the agent
may use, not a tool of its own, and registering it would make tool_manual resolve
'compile_scad' to a native tool that shadows the real MCP one.
"""
reg = ToolRegistry()
for name in sorted(NATIVE_TOOLS | MCP_FULL_NAMES):
reg.register(FakeTool(name), builtin=True)
return reg
# ── the manuals name real tools ──────────────────────────────────────────
def test_the_manuals_directory_is_not_empty():
"""Every assertion below is vacuous on an empty tree — fail loudly instead."""
assert _manuals(), f"no manuals found under {tool_manual.MANUALS_DIR}"
@pytest.mark.parametrize("path", _manuals(), ids=lambda p: p.stem)
def test_every_manual_is_named_after_a_tool_that_exists(path: Path):
"""This is what catches a manual for a tool nobody can call — write_tool.md was one."""
assert path.stem in KNOWN_TOOLS, (
f"manuals/{path.name} documents '{path.stem}', which is not a tool: no built-in or "
f"tools/*.py declares it, and no mcp_servers.d group lists it.{_closest(path.stem)}"
)
def test_a_guide_is_never_named_after_a_tool():
"""A guide sharing a tool's name would shadow that tool's manual in the index."""
for path in sorted(tool_manual.GUIDES_DIR.glob("*.md")):
assert path.stem not in KNOWN_TOOLS, (
f"manuals/guides/{path.name} is named after the real tool '{path.stem}' — a guide "
"is not a tool and must not be filed under a tool's name"
)
# ── the manuals are reachable under the names the agent uses ─────────────
@pytest.mark.parametrize("path", _manuals(), ids=lambda p: p.stem)
async def test_the_bare_name_returns_the_hand_written_manual(path: Path):
result = await ToolManualTool().execute({"tool_name": path.stem})
assert result.success is True
assert result.output == path.read_text(encoding="utf-8"), (
f"manuals/{path.name} exists but tool_manual('{path.stem}') did not return it"
)
@pytest.mark.parametrize(
"path", [p for p in _manuals() if _full_names_for(p.stem)], ids=lambda p: p.stem
)
async def test_an_mcp_manual_is_reachable_by_its_full_name(path: Path):
"""The regression, on the real file: the agent calls the MCP spelling, the file is
the bare one — so the call must reach the hand-written text, not the schema."""
for full_name in _full_names_for(path.stem):
result = await ToolManualTool().execute({"tool_name": full_name})
assert result.success is True
assert result.output == path.read_text(encoding="utf-8")
assert "not a hand-written manual" not in result.output
async def test_mcp_manuals_are_indexed_under_their_server(registry: ToolRegistry):
"""A manual is filed bare, so its source can only come from the registry."""
result = await ToolManualTool(registry=registry).execute({})
assert result.success is True
for path in _manuals():
for full_name in _full_names_for(path.stem):
server_source = full_name.rsplit("__", 1)[0] + "__"
assert f"{server_source} (" in result.output, (
f"manuals/{path.name} documents the tool '{full_name}', but the index does "
f"not file it under {server_source}"
)
def test_the_index_lists_every_manual_on_disk():
assert tool_manual.manual_names() == _manual_names_on_disk()
async def test_the_no_argument_call_succeeds_and_names_them(registry: ToolRegistry):
result = await ToolManualTool(registry=registry).execute({})
assert result.success is True
for path in _manuals():
assert path.stem in result.output
for guide in sorted(tool_manual.GUIDES_DIR.glob("*.md")):
assert guide.stem in result.output
# ── the docs and prompts cite manuals that exist ─────────────────────────
def _reference_sources() -> list[Path]:
"""Live docs and prompts. docs/archive/ is history, not a claim about today."""
sources = sorted((REPO_ROOT / "docs").glob("*.md"))
sources += sorted((REPO_ROOT / "navi/profiles").rglob("*.txt"))
sources += [REPO_ROOT / "persona_navi_code.txt", REPO_ROOT / "CLAUDE.md"]
return [path for path in sources if path.is_file()]
_TOOL_MANUAL_CALL = re.compile(r"""tool_manual\(\s*["']([^"']+)["']\s*\)""")
_MANUAL_PATH = re.compile(r"manuals/([A-Za-z0-9_./-]+\.md)")
def _citations(pattern: re.Pattern[str]) -> dict[str, list[str]]:
"""pattern's capture group -> ['<file>:<line>', …], over the live docs."""
found: dict[str, list[str]] = {}
for path in _reference_sources():
for number, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1):
for match in pattern.finditer(line):
found.setdefault(match.group(1), []).append(f"{path.relative_to(REPO_ROOT)}:{number}")
return found
def test_every_cited_manual_file_exists():
"""`manuals/<x>.md` in a doc is a promise the file is there — moving the guide
into guides/ broke exactly this, in docs/context_providers.md."""
missing = {
name: where
for name, where in _citations(_MANUAL_PATH).items()
if not (tool_manual.MANUALS_DIR / name).is_file()
}
assert not missing, f"cited manuals that do not exist: {missing}"
def test_every_tool_manual_call_resolves():
"""`tool_manual("x")` must reach something: a hand-written manual, a guide, a real
tool whose schema the tool can render, or a configured MCP server — naming a server
is how its instructions are read now that the system prompt only keeps one line.
A `<placeholder>` is documentation shorthand, not a name anyone will type.
"""
resolvable = KNOWN_TOOLS | set(MCP_BY_SERVER)
unresolved = {
name: where
for name, where in _citations(_TOOL_MANUAL_CALL).items()
if tool_manual._read_manual(name) is None
and name not in resolvable
and not name.startswith("<")
}
assert not unresolved, (
f"tool_manual() calls that resolve to nothing: {unresolved} — name a real tool, or "
"write the manual the docs promise"
)