Newer
Older
navi-1 / navi / tools / _internal / redact.py
"""Redaction of secret tool arguments before they reach the logs.

An ``ssh_exec`` password and a third-party MCP tool's ``api_key`` are the same
problem, so the rule is keyed on the *parameter name* and applied wherever
arguments are logged — the main loop, its background path, the sub-agent runner
and the logging middleware. Keying on names rather than on a per-tool
declaration means a new tool is covered the day it is written, and a tool that
forgot to declare nothing stays exposed.

A copy is returned, so the arguments handed to the tool and shipped to the
client are untouched: this is about what the log keeps, not about what the tool
needs.

Known gap: a secret embedded in a free-form string — ``terminal``'s ``command``,
``code_exec``'s ``code``, ``test_mcp_tool``'s ``arguments`` — is invisible to any
name-based rule and is logged as written. Nothing here can see inside a command
line. The only cure for that is not putting the secret in one.
"""

from __future__ import annotations

from typing import Any

# Matched against the whole key, case-insensitively, with '-' read as '_'.
# Masking a harmless field costs a little debugging detail; missing a real
# credential costs the credential, so the balance leans towards masking.
SENSITIVE_KEYS = frozenset(
    {
        "password",
        "passwd",
        "pwd",
        "secret",
        "client_secret",
        "secret_key",
        "token",
        "access_token",
        "refresh_token",
        "id_token",
        "auth_token",
        "api_key",
        "apikey",
        "private_key",
        "passphrase",
        "authorization",
        "credential",
        "credentials",
    }
)

# Qualified names (`db_password`, `user_token`) are as common as bare ones. Only
# unambiguous tails belong here: `max_tokens` and `token_count` must stay
# readable, so `token` alone is not a suffix.
SENSITIVE_SUFFIXES = (
    "_password",
    "_passwd",
    "_secret",
    "_token",
    "_api_key",
    "_apikey",
    "_private_key",
    "_passphrase",
)

REDACTED = "***"

# Arguments nest shallowly in practice (a dict, a list of dicts). The cap is a
# guard against a pathological self-referencing structure, not a real limit.
_MAX_DEPTH = 6


def is_sensitive_key(key: object) -> bool:
    """Whether a parameter name holds a credential."""
    if not isinstance(key, str):
        return False
    normalized = key.strip().lower().replace("-", "_")
    return normalized in SENSITIVE_KEYS or normalized.endswith(SENSITIVE_SUFFIXES)


def redact_args(args: Any, _depth: int = 0) -> Any:
    """Return a copy of *args* with every sensitive value replaced by ``***``.

    Recurses through dicts and lists so a password nested in a tool's
    ``arguments`` payload is masked too. Keys are kept: the name of a secret is
    not itself secret, and losing it makes the log unreadable.
    """
    if _depth >= _MAX_DEPTH:
        return args
    if isinstance(args, dict):
        return {
            key: REDACTED if is_sensitive_key(key) else redact_args(value, _depth + 1)
            for key, value in args.items()
        }
    if isinstance(args, (list, tuple)):
        return [redact_args(item, _depth + 1) for item in args]
    return args