feat(flr): make FLR session lifecycle visible and reapable (#3)
This commit was merged in pull request #3.
This commit is contained in:
+146
-17
@@ -15,8 +15,12 @@ from zerto_rewind_mcp.config import load_catalog, load_config
|
||||
from zerto_rewind_mcp.protection import find_from_rows
|
||||
from zerto_rewind_mcp.recover import (
|
||||
download_token_from,
|
||||
is_live_session,
|
||||
resolve_flr_path,
|
||||
session_id_from,
|
||||
session_id_of,
|
||||
session_rows,
|
||||
session_summary,
|
||||
wait_flr_ready,
|
||||
)
|
||||
from zerto_rewind_mcp.util import pick
|
||||
@@ -256,6 +260,51 @@ async def zerto_add_mutating_tool(
|
||||
return _dump({"ok": True, "entry": entry.as_dict()})
|
||||
|
||||
|
||||
async def _live_session_ids(client: ZertoClient) -> set[str]:
|
||||
try:
|
||||
rows = session_rows(await client.list_flrs())
|
||||
except ZertoError:
|
||||
return set()
|
||||
return {session_id_of(r) for r in rows if is_live_session(r) and session_id_of(r)}
|
||||
|
||||
|
||||
async def _teardown_flr(
|
||||
client: ZertoClient,
|
||||
session_id: str | None,
|
||||
before: set[str],
|
||||
) -> dict[str, Any]:
|
||||
"""Unmount the FLR session and say what happened.
|
||||
|
||||
A swallowed unmount failure is how a mount silently wedges the recovery
|
||||
site: FLR cannot run during clone/test/live/EJC, so a stuck session blocks
|
||||
the next recovery. Report it instead.
|
||||
|
||||
If session_id is None the start may still have succeeded on the ZVM while
|
||||
the response failed to parse, so reap anything new that appeared.
|
||||
"""
|
||||
out: dict[str, Any] = {"attempted": False, "ok": True, "ended": [], "failed": []}
|
||||
targets = [session_id] if session_id else []
|
||||
if not targets:
|
||||
orphans = sorted(await _live_session_ids(client) - before)
|
||||
targets = orphans
|
||||
out["orphans_reaped"] = orphans
|
||||
for sid in targets:
|
||||
out["attempted"] = True
|
||||
try:
|
||||
await client.end_flr(sid)
|
||||
out["ended"].append(sid)
|
||||
except ZertoError as exc:
|
||||
out["ok"] = False
|
||||
out["failed"].append({"session_id": sid, "message": str(exc)})
|
||||
if not out["ok"]:
|
||||
out["message"] = (
|
||||
"FLR session may still be mounted on the recovery site. "
|
||||
"List it with zerto_list_flr_sessions and end it with "
|
||||
"zerto_end_flr_session; a stuck mount blocks the next FLR."
|
||||
)
|
||||
return out
|
||||
|
||||
|
||||
@mcp.tool(
|
||||
name="zerto_recover_file",
|
||||
annotations={
|
||||
@@ -292,7 +341,9 @@ async def zerto_recover_file(
|
||||
client = get_client()
|
||||
dest = Path(dest_dir or _settings.get("recovery_dir") or "./recovered")
|
||||
dest.mkdir(parents=True, exist_ok=True)
|
||||
session_id = None
|
||||
session_id: str | None = None
|
||||
before = await _live_session_ids(client)
|
||||
payload: dict[str, Any]
|
||||
try:
|
||||
started = await client.start_flr(
|
||||
vpg_identifier,
|
||||
@@ -309,24 +360,102 @@ async def zerto_recover_file(
|
||||
name = Path(guest_path.replace("\\", "/")).name or "recovered.bin"
|
||||
out_path = dest / name
|
||||
out_path.write_bytes(blob)
|
||||
return _dump(
|
||||
{
|
||||
"ok": True,
|
||||
"path": str(out_path.resolve()),
|
||||
"bytes": len(blob),
|
||||
"session_id": session_id,
|
||||
"flr_path": flr_path,
|
||||
"message": f"Wrote {len(blob)} bytes to {out_path}",
|
||||
}
|
||||
)
|
||||
payload = {
|
||||
"ok": True,
|
||||
"path": str(out_path.resolve()),
|
||||
"bytes": len(blob),
|
||||
"session_id": session_id,
|
||||
"flr_path": flr_path,
|
||||
"message": f"Wrote {len(blob)} bytes to {out_path}",
|
||||
}
|
||||
except ZertoError as exc:
|
||||
payload = {"ok": False, "session_id": session_id, "message": str(exc)}
|
||||
finally:
|
||||
unmount = await _teardown_flr(client, session_id, before)
|
||||
payload["unmount"] = unmount
|
||||
if not unmount["ok"]:
|
||||
# The bytes are on disk, so ok stays true, but the caller must be told
|
||||
# the mount is still up rather than finding out at the next recovery.
|
||||
payload["warning"] = unmount["message"]
|
||||
return _dump(payload)
|
||||
|
||||
|
||||
@mcp.tool(
|
||||
name="zerto_list_flr_sessions",
|
||||
annotations={
|
||||
"title": "List FLR sessions on the ZVM",
|
||||
"readOnlyHint": True,
|
||||
"destructiveHint": False,
|
||||
"idempotentHint": True,
|
||||
"openWorldHint": True,
|
||||
},
|
||||
)
|
||||
async def zerto_list_flr_sessions(live_only: bool = True) -> str:
|
||||
"""Every FLR session the ZVM knows about, so orphaned mounts are visible.
|
||||
|
||||
zerto_recover_file tears its own session down, but that cleanup only runs if
|
||||
this process survives the call. A crash, disconnect or timeout mid-recovery
|
||||
leaves the mount up with nothing tracking it. FLR cannot run during clone,
|
||||
test, live failover or EJC, so a stuck mount blocks the next recovery.
|
||||
|
||||
live_only keeps sessions still holding a mount. Pass false to see ended and
|
||||
failed sessions too, which the ZVM keeps as history.
|
||||
"""
|
||||
try:
|
||||
rows = session_rows(await get_client().list_flrs())
|
||||
except ZertoError as exc:
|
||||
return _dump({"ok": False, "message": str(exc)})
|
||||
finally:
|
||||
if session_id:
|
||||
try:
|
||||
await client.end_flr(session_id)
|
||||
except ZertoError:
|
||||
pass
|
||||
kept = [r for r in rows if is_live_session(r)] if live_only else rows
|
||||
return _dump(
|
||||
{
|
||||
"ok": True,
|
||||
"live_only": live_only,
|
||||
"count": len(kept),
|
||||
"total_known": len(rows),
|
||||
"sessions": [session_summary(r) for r in kept],
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@mcp.tool(
|
||||
name="zerto_end_flr_session",
|
||||
annotations={
|
||||
"title": "End an FLR session (unmount)",
|
||||
"readOnlyHint": False,
|
||||
"destructiveHint": True,
|
||||
"idempotentHint": True,
|
||||
"openWorldHint": True,
|
||||
},
|
||||
)
|
||||
async def zerto_end_flr_session(session_id: str, confirmed: bool = False) -> str:
|
||||
"""Unmount an FLR session. Use to reap an orphan left by a crashed recovery.
|
||||
|
||||
Requires confirmed=true: ending a session that another operator is actively
|
||||
pulling files from will interrupt them. Find the id with
|
||||
zerto_list_flr_sessions.
|
||||
"""
|
||||
if not confirmed:
|
||||
return _dump(
|
||||
{
|
||||
"ok": False,
|
||||
"needs_confirm": True,
|
||||
"message": (
|
||||
"Set confirmed=true after a human yes. Ending a session that "
|
||||
"someone is actively recovering from will interrupt them."
|
||||
),
|
||||
}
|
||||
)
|
||||
try:
|
||||
await get_client().end_flr(session_id)
|
||||
except ZertoError as exc:
|
||||
return _dump({"ok": False, "session_id": session_id, "message": str(exc)})
|
||||
return _dump(
|
||||
{
|
||||
"ok": True,
|
||||
"session_id": session_id,
|
||||
"message": f"Ended FLR session {session_id}.",
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
@mcp.tool(
|
||||
|
||||
Reference in New Issue
Block a user