mirror of
https://github.com/langbot-app/LangBot.git
synced 2026-07-22 12:26:08 +00:00
fix(mcp): survive transient WS transport drops for Box stdio MCP servers (#2303)
* fix(mcp): survive transient WS transport drops for Box stdio MCP servers A Box-backed stdio MCP server (e.g. pab1it0/prometheus) would periodically error on the frontend with Box managed process exited unexpectedly / Failed after 4 attempts once the session had been alive for a while. Root cause: the managed MCP process lives in the Box runtime and SURVIVES a WebSocket transport drop, but _lifecycle_loop treated any monitor completion as a fatal process death. It then ran the finally-block cleanup — which STOPS the still-healthy managed process — and did a full 4-attempt exponential backoff rebuild. Under an occasionally-stalled single-worker event loop the mcp websocket client misses a ping/pong, the transport drops, and this self-inflicted teardown loop is what the user sees. Fixes: - _lifecycle_loop: when the health monitor completes, re-check the real managed-process state. If the process is still running, the transport merely dropped: raise an internal _TransportReconnect signal instead of Box managed process exited unexpectedly. - _lifecycle_loop_with_retry: handle _TransportReconnect as a free, uncounted reconnect (does not consume the fatal retry budget), so a long-lived session survives arbitrarily many transient drops. - finally-block: gate managed-process teardown on a _preserve_managed_process flag so a transport-only reconnect closes just the WS, not the process. - BoxStdioSessionRuntime.initialize: reuse an already-running managed process instead of stopping+rebuilding it (which also re-ran the slow dependency bootstrap); only (re)start when none is running. Adds _managed_process_is_running() helper. Pairs with langbot-plugin-sdk fix adding a server-driven WS heartbeat to the managed-process relay, which prevents most drops in the first place. * style(mcp): ruff format * chore(deps): pin langbot-plugin 0.4.8 for the managed-process WS heartbeat fix --------- Co-authored-by: dadachann <185672915+dadachann@users.noreply.github.com>
This commit is contained in:
@@ -173,25 +173,36 @@ class BoxStdioSessionRuntime:
|
||||
stderr_preview = (result.stderr or '')[:500]
|
||||
raise Exception(f'Dependency install failed (exit code {result.exit_code}): {stderr_preview}')
|
||||
|
||||
try:
|
||||
process_workspace = (
|
||||
self._build_workspace(host_path=host_path, workdir=process_cwd, mount_path=process_cwd)
|
||||
if host_path
|
||||
else workspace
|
||||
# Reuse an already-running managed process instead of rebuilding it.
|
||||
# The Box runtime keeps the managed process alive across a transient
|
||||
# WebSocket transport drop, so on a reconnect we only need to re-attach
|
||||
# the WS below. Rebuilding here would needlessly stop a healthy process
|
||||
# and re-run the (slow, network-touching) dependency bootstrap.
|
||||
if not await self._managed_process_is_running():
|
||||
try:
|
||||
process_workspace = (
|
||||
self._build_workspace(host_path=host_path, workdir=process_cwd, mount_path=process_cwd)
|
||||
if host_path
|
||||
else workspace
|
||||
)
|
||||
payload = process_workspace.build_process_payload(
|
||||
self.server_config['command'],
|
||||
self.server_config.get('args', []),
|
||||
env=self.server_config.get('env', {}),
|
||||
cwd=process_cwd,
|
||||
)
|
||||
if install_cmd:
|
||||
payload = self._wrap_process_payload_with_python_env(payload, process_cwd)
|
||||
payload['process_id'] = self.process_id
|
||||
await workspace.box_service.start_managed_process(workspace.session_id, payload)
|
||||
except Exception:
|
||||
self.owner.error_phase = MCPSessionErrorPhase.PROCESS_START
|
||||
raise
|
||||
else:
|
||||
self.ap.logger.info(
|
||||
f'MCP server {self.server_name}: reusing live managed process '
|
||||
f'process_id={self.process_id} (transport reconnect)'
|
||||
)
|
||||
payload = process_workspace.build_process_payload(
|
||||
self.server_config['command'],
|
||||
self.server_config.get('args', []),
|
||||
env=self.server_config.get('env', {}),
|
||||
cwd=process_cwd,
|
||||
)
|
||||
if install_cmd:
|
||||
payload = self._wrap_process_payload_with_python_env(payload, process_cwd)
|
||||
payload['process_id'] = self.process_id
|
||||
await workspace.box_service.start_managed_process(workspace.session_id, payload)
|
||||
except Exception:
|
||||
self.owner.error_phase = MCPSessionErrorPhase.PROCESS_START
|
||||
raise
|
||||
|
||||
try:
|
||||
websocket_url = workspace.get_managed_process_websocket_url(self.process_id)
|
||||
@@ -236,6 +247,23 @@ class BoxStdioSessionRuntime:
|
||||
return
|
||||
await asyncio.sleep(self.owner._MONITOR_POLL_INTERVAL)
|
||||
|
||||
async def _managed_process_is_running(self) -> bool:
|
||||
"""Return True if this server's managed process exists and is running.
|
||||
|
||||
Used to decide whether initialize() must (re)start the process or can
|
||||
simply re-attach the WebSocket transport to a process the Box runtime
|
||||
kept alive across a transient transport drop.
|
||||
"""
|
||||
from langbot_plugin.box.models import BoxManagedProcessStatus
|
||||
|
||||
workspace = self._build_workspace()
|
||||
try:
|
||||
info = await workspace.get_managed_process(self.process_id)
|
||||
except Exception:
|
||||
return False
|
||||
status = info.get('status', '') if isinstance(info, dict) else getattr(info, 'status', '')
|
||||
return status in (BoxManagedProcessStatus.RUNNING.value, BoxManagedProcessStatus.RUNNING)
|
||||
|
||||
async def _stage_host_path_to_shared_workspace(self, host_path: str) -> str:
|
||||
source_path = normalize_host_path(host_path)
|
||||
if not source_path:
|
||||
|
||||
Reference in New Issue
Block a user