feat(flr): make FLR session lifecycle visible and reapable

zerto_recover_file already tore its session down in a finally block, but
three gaps meant a mount could stay up on the recovery site with nothing
tracking it. FLR cannot run during clone, test, live failover or EJC, so
a stuck session blocks the next recovery.

1. An unmount failure was swallowed (`except ZertoError: pass`). The
   caller got ok=true and never learned the mount was still up. The
   teardown result is now reported in the response as `unmount`, with a
   `warning` when it fails. ok stays true when the bytes did land -- the
   recovery genuinely succeeded -- but the caller is told.

2. If start_flr succeeded on the ZVM while its response failed to parse,
   session_id stayed None and the finally block did nothing, leaking a
   session the process never knew the id of. Teardown now snapshots live
   session ids before starting and reaps anything new that appeared,
   leaving other operators' sessions alone.

3. Nothing could see or clear an orphan left by a crashed process, since
   the finally block only runs if the process survives. Two new tools:

   - zerto_list_flr_sessions: every session the ZVM knows about.
     live_only (default true) keeps the ones still holding a mount;
     ended and failed sessions linger as history and hold nothing.
   - zerto_end_flr_session: unmount one. Gated on confirmed=true,
     matching the other destructive tools, because ending a session
     someone else is pulling files from will interrupt them.

Verified against ZVM 10.x: listing reports 0 live / 1 known after a clean
run, the confirm gate refuses without a human yes, a real recovery from
checkpoint 1368 returned 158 bytes and reported
unmount.ok=true with the session id it ended, and 0 live sessions
remained afterwards.

pytest 27 passed (5 new, including fakes covering the swallowed-failure
and orphan-reap paths).

Co-Authored-By: Claude Opus 5 (1M context) <[email protected]>
Claude-Session: https://claude.ai/code/session_016yVfC5nvZowoLFnEGWhLGn
This commit is contained in:
2026-09-21 13:42:42 -04:00
co-authored by Claude Opus 5
parent 108e919dcb
commit 8b973653d8
7 changed files with 288 additions and 18 deletions
+146 -17
View File
@@ -15,8 +15,12 @@ from zerto_rewind_mcp.config import load_catalog, load_config
from zerto_rewind_mcp.protection import find_from_rows
from zerto_rewind_mcp.recover import (
download_token_from,
is_live_session,
resolve_flr_path,
session_id_from,
session_id_of,
session_rows,
session_summary,
wait_flr_ready,
)
from zerto_rewind_mcp.util import pick
@@ -256,6 +260,51 @@ async def zerto_add_mutating_tool(
return _dump({"ok": True, "entry": entry.as_dict()})
async def _live_session_ids(client: ZertoClient) -> set[str]:
try:
rows = session_rows(await client.list_flrs())
except ZertoError:
return set()
return {session_id_of(r) for r in rows if is_live_session(r) and session_id_of(r)}
async def _teardown_flr(
client: ZertoClient,
session_id: str | None,
before: set[str],
) -> dict[str, Any]:
"""Unmount the FLR session and say what happened.
A swallowed unmount failure is how a mount silently wedges the recovery
site: FLR cannot run during clone/test/live/EJC, so a stuck session blocks
the next recovery. Report it instead.
If session_id is None the start may still have succeeded on the ZVM while
the response failed to parse, so reap anything new that appeared.
"""
out: dict[str, Any] = {"attempted": False, "ok": True, "ended": [], "failed": []}
targets = [session_id] if session_id else []
if not targets:
orphans = sorted(await _live_session_ids(client) - before)
targets = orphans
out["orphans_reaped"] = orphans
for sid in targets:
out["attempted"] = True
try:
await client.end_flr(sid)
out["ended"].append(sid)
except ZertoError as exc:
out["ok"] = False
out["failed"].append({"session_id": sid, "message": str(exc)})
if not out["ok"]:
out["message"] = (
"FLR session may still be mounted on the recovery site. "
"List it with zerto_list_flr_sessions and end it with "
"zerto_end_flr_session; a stuck mount blocks the next FLR."
)
return out
@mcp.tool(
name="zerto_recover_file",
annotations={
@@ -292,7 +341,9 @@ async def zerto_recover_file(
client = get_client()
dest = Path(dest_dir or _settings.get("recovery_dir") or "./recovered")
dest.mkdir(parents=True, exist_ok=True)
session_id = None
session_id: str | None = None
before = await _live_session_ids(client)
payload: dict[str, Any]
try:
started = await client.start_flr(
vpg_identifier,
@@ -309,24 +360,102 @@ async def zerto_recover_file(
name = Path(guest_path.replace("\\", "/")).name or "recovered.bin"
out_path = dest / name
out_path.write_bytes(blob)
return _dump(
{
"ok": True,
"path": str(out_path.resolve()),
"bytes": len(blob),
"session_id": session_id,
"flr_path": flr_path,
"message": f"Wrote {len(blob)} bytes to {out_path}",
}
)
payload = {
"ok": True,
"path": str(out_path.resolve()),
"bytes": len(blob),
"session_id": session_id,
"flr_path": flr_path,
"message": f"Wrote {len(blob)} bytes to {out_path}",
}
except ZertoError as exc:
payload = {"ok": False, "session_id": session_id, "message": str(exc)}
finally:
unmount = await _teardown_flr(client, session_id, before)
payload["unmount"] = unmount
if not unmount["ok"]:
# The bytes are on disk, so ok stays true, but the caller must be told
# the mount is still up rather than finding out at the next recovery.
payload["warning"] = unmount["message"]
return _dump(payload)
@mcp.tool(
name="zerto_list_flr_sessions",
annotations={
"title": "List FLR sessions on the ZVM",
"readOnlyHint": True,
"destructiveHint": False,
"idempotentHint": True,
"openWorldHint": True,
},
)
async def zerto_list_flr_sessions(live_only: bool = True) -> str:
"""Every FLR session the ZVM knows about, so orphaned mounts are visible.
zerto_recover_file tears its own session down, but that cleanup only runs if
this process survives the call. A crash, disconnect or timeout mid-recovery
leaves the mount up with nothing tracking it. FLR cannot run during clone,
test, live failover or EJC, so a stuck mount blocks the next recovery.
live_only keeps sessions still holding a mount. Pass false to see ended and
failed sessions too, which the ZVM keeps as history.
"""
try:
rows = session_rows(await get_client().list_flrs())
except ZertoError as exc:
return _dump({"ok": False, "message": str(exc)})
finally:
if session_id:
try:
await client.end_flr(session_id)
except ZertoError:
pass
kept = [r for r in rows if is_live_session(r)] if live_only else rows
return _dump(
{
"ok": True,
"live_only": live_only,
"count": len(kept),
"total_known": len(rows),
"sessions": [session_summary(r) for r in kept],
}
)
@mcp.tool(
name="zerto_end_flr_session",
annotations={
"title": "End an FLR session (unmount)",
"readOnlyHint": False,
"destructiveHint": True,
"idempotentHint": True,
"openWorldHint": True,
},
)
async def zerto_end_flr_session(session_id: str, confirmed: bool = False) -> str:
"""Unmount an FLR session. Use to reap an orphan left by a crashed recovery.
Requires confirmed=true: ending a session that another operator is actively
pulling files from will interrupt them. Find the id with
zerto_list_flr_sessions.
"""
if not confirmed:
return _dump(
{
"ok": False,
"needs_confirm": True,
"message": (
"Set confirmed=true after a human yes. Ending a session that "
"someone is actively recovering from will interrupt them."
),
}
)
try:
await get_client().end_flr(session_id)
except ZertoError as exc:
return _dump({"ok": False, "session_id": session_id, "message": str(exc)})
return _dump(
{
"ok": True,
"session_id": session_id,
"message": f"Ended FLR session {session_id}.",
}
)
@mcp.tool(