wait_for_tag gave up after a hardcoded 45s. Checkpoint cadence is set by the VPG's protected site and the spread is enormous: 5s on vSphere, 60s on Azure, 630s on AWS, measured in one estate. A single constant cannot serve all three. Generous for vSphere, where a tag surfaces in about 4s. Impossible for AWS, where it takes ~128s. So the guard reported "no checkpoint, refusing the change" while Zerto was in the middle of creating one, and the checkpoint landed a minute after the agent had been told there was no rewind point. That is worse than the failure it guards against. It is silent, it reads as correct in the log, and it blocks legitimate work on every cloud-protected VM in the estate. The wait is now measured: cadence_seconds() takes the median gap of recent checkpoints, tag_wait_budget() turns that into 2x cadence plus headroom, clamped to 45s..300s, and scales the poll interval with it so a 630s VPG is not polled every 1.5s. Visibility does not scale linearly with cadence, because the insert creates its own off-cadence checkpoint, which is why this is a bounded multiple rather than a proportion. wait_for_tag also checks once before sleeping, so an already-present tag returns without a poll cycle. The failure message now names the measured cadence and says a completed Zerto task with no visible checkpoint means the wait was short, not that the insert was rejected. That was the exact wrong conclusion the old message invited. Hook budgets follow: too small a budget there just relocates the false denial from the guard into the hook, since a cancelled hook has its output discarded and the call proceeds unguarded. ZERTO_HOOK_GUARD_TIMEOUT 150 -> 330, settings timeout 180 -> 360, both above the 300s cap. Verified live against ZVM 10.9.10. Guard on win2019-1 (VPG CMH-AWS-1, AWS-protected) previously denied at 63s; now succeeds in 110.6s with checkpoint 186. jp-ubuntu unchanged at 6.6s, so vSphere pays nothing for this. Through the hook itself: win2019-1 allowed in 111s with cp 187, jp-ubuntu allowed in 9s with cp 23198. pytest 69 passed (6 new, including one asserting the budget exceeds the visibility actually measured on each of the three platforms). Co-Authored-By: Claude Opus 5 (1M context) <[email protected]> Claude-Session: https://claude.ai/code/session_016yVfC5nvZowoLFnEGWhLGn
252 lines
9.3 KiB
Python
Executable File
252 lines
9.3 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Claude Code PreToolUse hook: capture before execute, enforced by the host.
|
|
|
|
The MCP server cannot see another server's tool calls, so the guard it exposes
|
|
is advice the model may skip. This hook sits in the host, where the call really
|
|
does pause, so a failed checkpoint stops the change instead of merely
|
|
suggesting it should.
|
|
|
|
Decisions:
|
|
|
|
read-only tool allow, no checkpoint
|
|
mutating, guard ok allow, and tell the model which checkpoint to use
|
|
mutating, guard failed DENY. No checkpoint means no rewind, so no change.
|
|
mutating, VM unprotected prompt. Zerto cannot rewind it; a human decides.
|
|
unknown tool prompt. Nobody said it was read-only.
|
|
|
|
Exit 0 always, with the decision in stdout JSON. Exit 2 would block
|
|
unconditionally and ignore the JSON, which loses the reason text.
|
|
|
|
Install (project .claude/settings.json). Give it room: the guard inserts a
|
|
tagged checkpoint and waits for the Zerto task to reach Completed.
|
|
|
|
{"hooks": {"PreToolUse": [{
|
|
"matcher": "mcp__.*",
|
|
"hooks": [{"type": "command",
|
|
"command": "/path/to/.venv/bin/python /path/to/hooks/zerto_guard_hook.py",
|
|
"timeout": 360}]}]}}
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
import time
|
|
|
|
LOG = os.environ.get("ZERTO_HOOK_LOG", os.path.expanduser("~/.zerto-guard-hook.log"))
|
|
# Seconds the hook will wait for the checkpoint. Must stay under the hook
|
|
# timeout configured in settings.json, or the host cancels us and the tool
|
|
# call proceeds unguarded through the normal permission flow.
|
|
# Must exceed the largest tag wait the guard can take. That is now derived from
|
|
# the VPG's checkpoint cadence and capped at 300s (MAX_TAG_TIMEOUT_S), because a
|
|
# tag takes ~128s to surface on an AWS-protected VPG. Too small a budget here
|
|
# just moves the false denial from the guard into the hook.
|
|
GUARD_TIMEOUT_S = float(os.environ.get("ZERTO_HOOK_GUARD_TIMEOUT", "330"))
|
|
UNKNOWN_DECISION = os.environ.get("ZERTO_HOOK_UNKNOWN", "prompt") # prompt | allow | deny
|
|
|
|
|
|
def log(msg: str) -> None:
|
|
try:
|
|
with open(LOG, "a") as fh:
|
|
fh.write(f"{time.strftime('%Y-%m-%dT%H:%M:%S')} {msg}\n")
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def emit(decision: str | None, reason: str = "", context: str = "") -> None:
|
|
"""Write the PreToolUse decision and leave. Never raises."""
|
|
if decision is None and not context:
|
|
sys.exit(0) # no opinion; normal permission flow applies
|
|
out: dict = {"hookEventName": "PreToolUse"}
|
|
if decision:
|
|
out["permissionDecision"] = decision
|
|
out["permissionDecisionReason"] = reason
|
|
if context:
|
|
out["additionalContext"] = context
|
|
print(json.dumps({"hookSpecificOutput": out}))
|
|
sys.exit(0)
|
|
|
|
|
|
def split_mcp_name(tool_name: str) -> tuple[str, str] | None:
|
|
"""mcp__<server>__<tool> -> (server, tool). Server names may contain _."""
|
|
m = re.match(r"^mcp__(.+?)__(.+)$", tool_name or "")
|
|
return (m.group(1), m.group(2)) if m else None
|
|
|
|
|
|
def resolve_vm(tool_input: dict, vm_arg: str) -> str:
|
|
"""Pull the VM identifier out of the pending call's arguments."""
|
|
if vm_arg and tool_input.get(vm_arg):
|
|
return str(tool_input[vm_arg])
|
|
# vm_arg is the catalog's answer, but fall back to the usual suspects so a
|
|
# slightly-wrong catalog entry degrades to a prompt rather than a crash.
|
|
for key in (
|
|
"host",
|
|
"hostname",
|
|
"computer_name",
|
|
"computerName",
|
|
"vm",
|
|
"vm_name",
|
|
"target",
|
|
"limit",
|
|
"address",
|
|
"server",
|
|
):
|
|
if tool_input.get(key):
|
|
return str(tool_input[key])
|
|
return ""
|
|
|
|
|
|
async def run_guard(query: str, change_id: str, action: str) -> dict:
|
|
from zerto_rewind_mcp.checkpoints import make_tag, tag_vpgs
|
|
from zerto_rewind_mcp.client import ZertoClient, ZertoError
|
|
from zerto_rewind_mcp.config import load_config
|
|
from zerto_rewind_mcp.protection import find_from_rows
|
|
|
|
settings = load_config()
|
|
client = ZertoClient(
|
|
base_url=str(settings["zerto_url"]),
|
|
username=str(settings["username"]),
|
|
password=str(settings.get("password") or ""),
|
|
client_id=str(settings.get("client_id") or "zerto-client"),
|
|
verify_tls=bool(settings.get("verify_tls")),
|
|
)
|
|
try:
|
|
# Two different failures hide here. "Zerto has never heard of this VM"
|
|
# is not the same as "Zerto refused to tag it", and they get different
|
|
# answers, so a lookup miss must not surface as a refusal. Looking up a
|
|
# plain hostname by vmIdentifier returns HTTP 400, which is a miss.
|
|
rows = []
|
|
for kwargs in ({"vm_name": query}, {"vm_identifier": query}):
|
|
try:
|
|
rows = await client.get_vms(**kwargs)
|
|
except ZertoError:
|
|
rows = []
|
|
if rows:
|
|
break
|
|
if not rows:
|
|
return {
|
|
"ok": False,
|
|
"protected": False,
|
|
"message": f"Zerto has no VM matching {query!r}.",
|
|
}
|
|
result = find_from_rows(query, rows)
|
|
if result.outcome != "ok" or not result.taggable_vpgs:
|
|
return {"ok": False, "protected": False, "message": result.message}
|
|
tag = make_tag(
|
|
"claude-code",
|
|
change_id,
|
|
action=action,
|
|
vm_name=result.vm.vm_name if result.vm else None,
|
|
)
|
|
out = await tag_vpgs(client, result, tag)
|
|
out["protected"] = True
|
|
return out
|
|
except ZertoError as exc:
|
|
# Reached Zerto and it refused, or never reached it at all. Either way
|
|
# there is no checkpoint, so this is not a pass.
|
|
return {"ok": False, "protected": True, "message": str(exc)}
|
|
finally:
|
|
try:
|
|
await client.aclose()
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def main() -> None:
|
|
try:
|
|
payload = json.loads(sys.stdin.read() or "{}")
|
|
except Exception as exc:
|
|
log(f"unparseable stdin: {exc}")
|
|
emit(None)
|
|
|
|
tool_name = payload.get("tool_name") or ""
|
|
tool_input = payload.get("tool_input") or {}
|
|
parts = split_mcp_name(tool_name)
|
|
if not parts:
|
|
emit(None) # not an MCP tool; this hook has nothing to say
|
|
server, tool = parts
|
|
|
|
try:
|
|
from zerto_rewind_mcp.catalog import MUTATING, READ_ONLY
|
|
from zerto_rewind_mcp.config import load_catalog, load_config
|
|
|
|
catalog = load_catalog(load_config())
|
|
verdict = catalog.classify(server, tool)
|
|
entry = catalog.get(server, tool)
|
|
except Exception as exc:
|
|
log(f"catalog unavailable ({exc}); staying out of the way")
|
|
emit(None)
|
|
|
|
if verdict == READ_ONLY:
|
|
log(f"{tool_name}: read-only, allowed")
|
|
emit(None)
|
|
|
|
if verdict != MUTATING:
|
|
log(f"{tool_name}: unknown -> {UNKNOWN_DECISION}")
|
|
emit(
|
|
UNKNOWN_DECISION,
|
|
f"{server}/{tool} is not a known read-only tool and is not in the "
|
|
"Zerto mutating catalog, so no checkpoint was taken. If it changes a "
|
|
"protected VM, that change cannot be rewound. Allow it?",
|
|
)
|
|
|
|
vm = resolve_vm(tool_input, entry.vm_arg if entry else "")
|
|
if not vm:
|
|
log(f"{tool_name}: mutating but no VM in args {sorted(tool_input)}")
|
|
emit(
|
|
"prompt",
|
|
f"{server}/{tool} is a guest-mutating tool, but no VM could be read "
|
|
f"from its arguments (expected {entry.vm_arg if entry else '?'}). "
|
|
"No checkpoint was taken. Allow it?",
|
|
)
|
|
|
|
change_id = f"{payload.get('session_id', 'session')[:8]}-{payload.get('tool_use_id', '')[-8:]}"
|
|
action = f"{server}/{tool} on {vm}"
|
|
try:
|
|
guard = asyncio.run(asyncio.wait_for(run_guard(vm, change_id, action), GUARD_TIMEOUT_S))
|
|
except TimeoutError:
|
|
log(f"{tool_name}: guard timed out after {GUARD_TIMEOUT_S}s -> deny")
|
|
emit(
|
|
"deny",
|
|
f"Zerto did not confirm a tagged checkpoint for {vm} within "
|
|
f"{GUARD_TIMEOUT_S:.0f}s, so the change cannot be rewound. Refusing.",
|
|
)
|
|
except Exception as exc:
|
|
log(f"{tool_name}: guard raised {type(exc).__name__}: {exc} -> deny")
|
|
emit("deny", f"The Zerto guard failed for {vm}: {exc}. No checkpoint, so refusing.")
|
|
|
|
if guard.get("ok"):
|
|
tagged = (guard.get("tagged") or [{}])[0]
|
|
log(f"{tool_name}: tagged cp {tagged.get('checkpoint_id')} on {vm}, allowed")
|
|
emit(
|
|
None,
|
|
context=(
|
|
f"Zerto tagged checkpoint {tagged.get('checkpoint_id')} on VPG "
|
|
f"{tagged.get('vpg_name')} for {vm} before this call, tag "
|
|
f"{guard.get('tag')!r}. If this change breaks the guest, recover from "
|
|
"that checkpoint with zerto_recover_file."
|
|
),
|
|
)
|
|
|
|
if not guard.get("protected", True):
|
|
log(f"{tool_name}: {vm} not protected -> prompt")
|
|
emit(
|
|
"prompt",
|
|
f"{vm} is not protected by Zerto, so this change cannot be rewound. "
|
|
f"({guard.get('message', '')}) Allow it anyway?",
|
|
)
|
|
|
|
log(f"{tool_name}: guard failed on {vm} -> deny :: {guard.get('message')}")
|
|
emit(
|
|
"deny",
|
|
f"Zerto could not tag a checkpoint for {vm}, so this change would not be "
|
|
f"recoverable: {guard.get('message')}",
|
|
)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|