feat(flr): make FLR session lifecycle visible and reapable (#3)

This commit was merged in pull request #3.
This commit is contained in:
2026-09-21 13:43:05 -04:00
parent 108e919dcb
commit 59d1f71617
7 changed files with 288 additions and 18 deletions
+146 -17
View File
@@ -15,8 +15,12 @@ from zerto_rewind_mcp.config import load_catalog, load_config
from zerto_rewind_mcp.protection import find_from_rows
from zerto_rewind_mcp.recover import (
download_token_from,
is_live_session,
resolve_flr_path,
session_id_from,
session_id_of,
session_rows,
session_summary,
wait_flr_ready,
)
from zerto_rewind_mcp.util import pick
@@ -256,6 +260,51 @@ async def zerto_add_mutating_tool(
return _dump({"ok": True, "entry": entry.as_dict()})
async def _live_session_ids(client: ZertoClient) -> set[str]:
try:
rows = session_rows(await client.list_flrs())
except ZertoError:
return set()
return {session_id_of(r) for r in rows if is_live_session(r) and session_id_of(r)}
async def _teardown_flr(
client: ZertoClient,
session_id: str | None,
before: set[str],
) -> dict[str, Any]:
"""Unmount the FLR session and say what happened.
A swallowed unmount failure is how a mount silently wedges the recovery
site: FLR cannot run during clone/test/live/EJC, so a stuck session blocks
the next recovery. Report it instead.
If session_id is None the start may still have succeeded on the ZVM while
the response failed to parse, so reap anything new that appeared.
"""
out: dict[str, Any] = {"attempted": False, "ok": True, "ended": [], "failed": []}
targets = [session_id] if session_id else []
if not targets:
orphans = sorted(await _live_session_ids(client) - before)
targets = orphans
out["orphans_reaped"] = orphans
for sid in targets:
out["attempted"] = True
try:
await client.end_flr(sid)
out["ended"].append(sid)
except ZertoError as exc:
out["ok"] = False
out["failed"].append({"session_id": sid, "message": str(exc)})
if not out["ok"]:
out["message"] = (
"FLR session may still be mounted on the recovery site. "
"List it with zerto_list_flr_sessions and end it with "
"zerto_end_flr_session; a stuck mount blocks the next FLR."
)
return out
@mcp.tool(
name="zerto_recover_file",
annotations={
@@ -292,7 +341,9 @@ async def zerto_recover_file(
client = get_client()
dest = Path(dest_dir or _settings.get("recovery_dir") or "./recovered")
dest.mkdir(parents=True, exist_ok=True)
session_id = None
session_id: str | None = None
before = await _live_session_ids(client)
payload: dict[str, Any]
try:
started = await client.start_flr(
vpg_identifier,
@@ -309,24 +360,102 @@ async def zerto_recover_file(
name = Path(guest_path.replace("\\", "/")).name or "recovered.bin"
out_path = dest / name
out_path.write_bytes(blob)
return _dump(
{
"ok": True,
"path": str(out_path.resolve()),
"bytes": len(blob),
"session_id": session_id,
"flr_path": flr_path,
"message": f"Wrote {len(blob)} bytes to {out_path}",
}
)
payload = {
"ok": True,
"path": str(out_path.resolve()),
"bytes": len(blob),
"session_id": session_id,
"flr_path": flr_path,
"message": f"Wrote {len(blob)} bytes to {out_path}",
}
except ZertoError as exc:
payload = {"ok": False, "session_id": session_id, "message": str(exc)}
finally:
unmount = await _teardown_flr(client, session_id, before)
payload["unmount"] = unmount
if not unmount["ok"]:
# The bytes are on disk, so ok stays true, but the caller must be told
# the mount is still up rather than finding out at the next recovery.
payload["warning"] = unmount["message"]
return _dump(payload)
@mcp.tool(
name="zerto_list_flr_sessions",
annotations={
"title": "List FLR sessions on the ZVM",
"readOnlyHint": True,
"destructiveHint": False,
"idempotentHint": True,
"openWorldHint": True,
},
)
async def zerto_list_flr_sessions(live_only: bool = True) -> str:
"""Every FLR session the ZVM knows about, so orphaned mounts are visible.
zerto_recover_file tears its own session down, but that cleanup only runs if
this process survives the call. A crash, disconnect or timeout mid-recovery
leaves the mount up with nothing tracking it. FLR cannot run during clone,
test, live failover or EJC, so a stuck mount blocks the next recovery.
live_only keeps sessions still holding a mount. Pass false to see ended and
failed sessions too, which the ZVM keeps as history.
"""
try:
rows = session_rows(await get_client().list_flrs())
except ZertoError as exc:
return _dump({"ok": False, "message": str(exc)})
finally:
if session_id:
try:
await client.end_flr(session_id)
except ZertoError:
pass
kept = [r for r in rows if is_live_session(r)] if live_only else rows
return _dump(
{
"ok": True,
"live_only": live_only,
"count": len(kept),
"total_known": len(rows),
"sessions": [session_summary(r) for r in kept],
}
)
@mcp.tool(
name="zerto_end_flr_session",
annotations={
"title": "End an FLR session (unmount)",
"readOnlyHint": False,
"destructiveHint": True,
"idempotentHint": True,
"openWorldHint": True,
},
)
async def zerto_end_flr_session(session_id: str, confirmed: bool = False) -> str:
"""Unmount an FLR session. Use to reap an orphan left by a crashed recovery.
Requires confirmed=true: ending a session that another operator is actively
pulling files from will interrupt them. Find the id with
zerto_list_flr_sessions.
"""
if not confirmed:
return _dump(
{
"ok": False,
"needs_confirm": True,
"message": (
"Set confirmed=true after a human yes. Ending a session that "
"someone is actively recovering from will interrupt them."
),
}
)
try:
await get_client().end_flr(session_id)
except ZertoError as exc:
return _dump({"ok": False, "session_id": session_id, "message": str(exc)})
return _dump(
{
"ok": True,
"session_id": session_id,
"message": f"Ended FLR session {session_id}.",
}
)
@mcp.tool(