Skip to content
Merged
Show file tree
Hide file tree
Changes from 55 commits
Commits
Show all changes
58 commits
Select commit Hold shift + click to select a range
24bf043
fix(stop): tear down the dashboard port-forward on stop (#7227)
yanyunl1991 Jul 20, 2026
e7fa323
fix(stop): release the dashboard forward on idempotent (already-stopp…
yanyunl1991 Jul 20, 2026
3934974
fix: preserve sandbox recovery readiness guard
scarab-systems Jul 21, 2026
377ad1d
fix: preserve relaunch managed health retries
scarab-systems Jul 21, 2026
cf27925
test: keep relaunch regression linear
scarab-systems Jul 21, 2026
720342d
test(stop): cover sandbox-qualified forward teardown
cjagwani Jul 22, 2026
5da33a4
Merge branch 'main' into fix/7227-stop-teardown-dashboard-forward
prekshivyas Jul 22, 2026
3744f7f
Merge branch 'main' into fix/7227-stop-teardown-dashboard-forward
prekshivyas Jul 23, 2026
d9d26df
docs(stop): document dashboard forward cleanup
prekshivyas Jul 23, 2026
3eb5fb5
test(stop): cover partial stop cleanup suppression
prekshivyas Jul 23, 2026
475264b
fix(stop): keep forward cleanup best effort
prekshivyas Jul 23, 2026
2674a52
fix(stop): harden dashboard forward cleanup
prekshivyas Jul 23, 2026
6f803a5
test(stop): cover failed forward cleanup
prekshivyas Jul 23, 2026
6fd6e69
Merge remote-tracking branch 'origin/main' into codex/pr-7228-finish
prekshivyas Jul 23, 2026
efca483
test(e2e): cover scoped dashboard forward stop
prekshivyas Jul 23, 2026
17daba5
test(e2e): reuse dashboard forward polling
prekshivyas Jul 23, 2026
0f3ab44
Merge branch 'main' into fix/7227-stop-teardown-dashboard-forward
prekshivyas Jul 23, 2026
9308428
Merge branch 'main' into fix/7227-stop-teardown-dashboard-forward
prekshivyas Jul 23, 2026
ebdbed6
Merge branch 'main' into fix/7227-stop-teardown-dashboard-forward
prekshivyas Jul 23, 2026
c5e1c2d
test(e2e): scope dashboard forward stop coverage
prekshivyas Jul 23, 2026
8582ff8
Merge remote-tracking branch 'origin/main' into codex/nvqa-7228-refre…
prekshivyas Jul 23, 2026
4966110
Merge remote-tracking branch 'origin/pull-request/7291' into codex/nv…
prekshivyas Jul 23, 2026
1dbed9f
test(e2e): restore dashboard restart regression
prekshivyas Jul 23, 2026
92433a3
test(e2e): fail closed on forward inspection
prekshivyas Jul 24, 2026
0097e77
Merge remote-tracking branch 'origin/main' into codex/nvqa-7228-refre…
prekshivyas Jul 24, 2026
fca2054
Merge remote-tracking branch 'origin/fix/7227-stop-teardown-dashboard…
prekshivyas Jul 24, 2026
4f4bd81
chore(cli): sort forward recovery imports
prekshivyas Jul 24, 2026
87608f8
fix(cli): wait for OpenShell after sandbox restart
prekshivyas Jul 24, 2026
4a00e47
fix(recovery): preserve sandbox readiness timeout
prekshivyas Jul 24, 2026
41c676e
merge(main): refresh PR #7228
prekshivyas Jul 24, 2026
eea89ba
fix(onboard): restore dashboard forward after recovery
prekshivyas Jul 24, 2026
96a478a
chore(onboard): keep entrypoint net-neutral
prekshivyas Jul 24, 2026
cb336b6
Merge branch 'main' into fix/7227-stop-teardown-dashboard-forward
prekshivyas Jul 24, 2026
42e186a
Merge branch 'main' into fix/7227-stop-teardown-dashboard-forward
prekshivyas Jul 24, 2026
47297b3
docs(recovery): scope readiness behavior by agent
prekshivyas Jul 24, 2026
8324b3a
Merge branch 'main' into fix/7227-stop-teardown-dashboard-forward
prekshivyas Jul 24, 2026
b100e41
merge(main): refresh PR #7228
prekshivyas Jul 24, 2026
c983284
Merge branch 'main' into fix/7227-stop-teardown-dashboard-forward
prekshivyas Jul 24, 2026
1c06200
Merge branch 'main' into fix/7227-stop-teardown-dashboard-forward
prekshivyas Jul 24, 2026
101e70f
Merge branch 'main' into fix/7227-stop-teardown-dashboard-forward
prekshivyas Jul 24, 2026
1732043
Merge branch 'main' into fix/7227-stop-teardown-dashboard-forward
prekshivyas Jul 24, 2026
cd5160e
fix(sandbox): retry supervisor reconnect during recovery
prekshivyas Jul 24, 2026
bf99111
fix(recovery): retry timed-out readiness probe
prekshivyas Jul 24, 2026
06526e7
merge(main): refresh PR #7228
prekshivyas Jul 24, 2026
848069e
docs(recovery): record reconnect retry removal condition
prekshivyas Jul 24, 2026
1f42879
Merge remote-tracking branch 'origin/main' into codex/pr-7228-refresh…
prekshivyas Jul 24, 2026
166787e
fix(recovery): honor sandbox readiness budget
prekshivyas Jul 24, 2026
022be54
Merge remote-tracking branch 'origin/main' into codex/pr-7228-refresh…
prekshivyas Jul 24, 2026
0e06140
fix(onboard): wait after policy application
prekshivyas Jul 24, 2026
7bfef2c
fix(onboard): verify control plane after policy apply
prekshivyas Jul 24, 2026
8c6f125
docs(onboard): document post-policy readiness
prekshivyas Jul 24, 2026
51651d0
test(onboard): cover post-policy readiness failure
prekshivyas Jul 24, 2026
198352b
fix(sandbox): retry OpenShell relay readiness
prekshivyas Jul 24, 2026
7d1075e
merge(main): refresh PR #7228
prekshivyas Jul 24, 2026
14387c7
fix(onboard): reconcile final recovery forward
prekshivyas Jul 24, 2026
f7acc80
fix(sandbox): retry nested relay timeout
prekshivyas Jul 24, 2026
7239d91
Merge branch 'main' into fix/7227-stop-teardown-dashboard-forward
prekshivyas Jul 24, 2026
ba010ec
fix(sandbox): retry relay target startup during recovery
jyaunches Jul 24, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 14 additions & 1 deletion docs/inference/configure-inference-timeouts.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -20,7 +20,7 @@ Use the error location to select the correct setting.
|---|---|---|
| `NEMOCLAW_AGENT_TIMEOUT` | OpenClaw per-request inference | `600` seconds |
| `NEMOCLAW_LOCAL_INFERENCE_TIMEOUT` | Ollama, vLLM, NIM, and compatible-endpoint validation during onboarding | `180` seconds |
| `NEMOCLAW_SANDBOX_READY_TIMEOUT` | Image build, gateway upload, and in-sandbox boot after sandbox creation | `180` seconds |
| `NEMOCLAW_SANDBOX_READY_TIMEOUT` | Image build, gateway upload, and in-sandbox boot after creation; OpenShell command re-registration after policy application or after OpenClaw or Hermes managed recovery recreates the sandbox | `180` seconds |

The readiness timeout does not govern inference requests or provider validation.

Expand Down Expand Up @@ -64,12 +64,25 @@ This variable does not extend the later sandbox-readiness wait.

Raise `NEMOCLAW_SANDBOX_READY_TIMEOUT` when onboarding creates the sandbox but image build, upload, or boot exceeds 180 seconds.
This can occur during a first run with cold caches or on a remote VM over a slow link.
Onboarding also uses this budget to confirm that the sandbox can execute commands again after applying policy presets.

<AgentOnly variant="openclaw,hermes">

The same budget applies when `start` or `recover` transactionally recreates a managed sandbox and waits for OpenShell to re-register it.

</AgentOnly>

```bash
export NEMOCLAW_SANDBOX_READY_TIMEOUT=600
$$nemoclaw onboard
```

<AgentOnly variant="openclaw,hermes">

For an existing sandbox, export the variable before the `start` or `recover` command that performs the recreation.

</AgentOnly>

Raise both onboarding budgets when the provider probe and the later sandbox creation phase are slow.

```bash
Expand Down
2 changes: 2 additions & 0 deletions docs/manage-sandboxes/recover-rebuild-sandboxes.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -84,6 +84,8 @@ For a local Docker-driver sandbox whose container still uses the legacy keepaliv

NemoClaw keeps the previous container available until the managed controller proves the supervisor topology, gateway health, and settle check, and attempts to restore it if that proof fails.
The recreation preserves mounted sandbox state, but a committed swap does not retain changes stored only in the previous container's writable layer.
After a transactional recreation, NemoClaw uses the `NEMOCLAW_SANDBOX_READY_TIMEOUT` budget (180 seconds by default) for OpenShell to re-register the sandbox before starting the primary dashboard or API host forward.
A definitive managed-health failure still stops immediately; if re-registration does not complete within the budget, the forward stays stopped.

For the controller topology, trust boundary, and fail-closed conditions, refer to [Understand Gateway Lifecycle Control](../configure-sandboxes/understand-gateway-lifecycle-control).
If recovery cannot repair a sandbox that needs credentials or a current controller contract, rebuild it.
Expand Down
4 changes: 3 additions & 1 deletion docs/manage-sandboxes/run-sandboxes.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -118,8 +118,10 @@ Stop a sandbox's container to free CPU, memory, and GPU resources without losing
$$nemoclaw <sandbox-name> stop
```

Workspace files, credentials, network policies, and the registry entry are preserved; only the container stops running.
Workspace files, credentials, network policies, and the registry entry are preserved.
The container stops running.
<AgentOnly variant="openclaw,hermes">
After the container stops, NemoClaw attempts to stop that sandbox's host dashboard forward.
The shared host gateway and tunnel services keep serving other sandboxes.
</AgentOnly>

Expand Down
32 changes: 27 additions & 5 deletions docs/reference/commands.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -1186,7 +1186,7 @@ Terminal agents do not have a gateway runtime and fail as unsupported.
### `$$nemoclaw <name> stop`

Stop the sandbox's Docker container while preserving all of its state.
Workspace files, credentials, network policies, the registry entry, and the OpenShell sandbox record stay in place; only the container stops running.
Workspace files, credentials, network policies, the registry entry, and the OpenShell sandbox record stay in place.
Use this to free CPU, memory, and GPU resources without destroying the sandbox; use [`$$nemoclaw <name> destroy`](#$$nemoclaw-name-destroy) when you want to delete it instead.

```bash
Expand All @@ -1195,8 +1195,17 @@ $$nemoclaw my-assistant stop

For OpenClaw-managed gateways, the command first asks the in-sandbox gateway to shut down its channels gracefully; agent-managed gateways (for example Hermes) are supervised inside the sandbox and shut down with the container's stop signal.
Then the container stops; a container stuck in a crash loop is stopped the same way, which also disarms its restart policy.
<AgentOnly variant="openclaw,hermes">
After the container stops, NemoClaw attempts to stop that sandbox's host dashboard forward.
If the container does not stop, NemoClaw leaves the dashboard forward running.
</AgentOnly>
The shared host gateway, tunnel services, and any local NIM inference container serve other sandboxes and keep running.
<AgentOnly variant="openclaw,hermes">
Stopping an already-stopped sandbox succeeds and attempts to remove any leftover dashboard forward for that sandbox.
</AgentOnly>
<AgentOnly variant="deepagents">
Stopping an already-stopped sandbox succeeds without changes.
</AgentOnly>
The command controls the local container directly, so it is available for local-container drivers (the default Docker driver and the vm driver) and unavailable for remote drivers such as kubernetes; if the Docker daemon itself is unreachable, the command reports the outage instead of guessing at container state.

### `$$nemoclaw <name> start`
Expand Down Expand Up @@ -3663,27 +3672,40 @@ Defaults are sized for typical hardware; override only if you see false-positive
| `NEMOCLAW_SANDBOX_EXEC_TIMEOUT_MS` | per call site (typically `15000`) | Overrides the default timeout for `openshell sandbox exec` calls issued by recovery and lifecycle helpers. Integer milliseconds; non-positive or non-numeric values fall back to the per-call-site default. |
| `NEMOCLAW_STATUS_PROBE_TIMEOUT_MS` | built-in default | Overrides the timeout for the OpenShell status probe used by `$$nemoclaw <name> status`. Integer milliseconds; non-positive or non-numeric values fall back to the default. |

### Onboard Timeouts
### Onboard and Sandbox Readiness Timeouts

The following environment variables tune onboard-time wall-clock limits.
`NEMOCLAW_SANDBOX_READY_TIMEOUT` also covers OpenShell command re-registration after onboarding applies policy presets.

<AgentOnly variant="openclaw,hermes">
`NEMOCLAW_SANDBOX_READY_TIMEOUT` also applies when managed recovery transactionally recreates an existing sandbox.
</AgentOnly>
Set them before running `$$nemoclaw onboard` if a slow connection or large model pull risks tripping the default.

| Variable | Default | Purpose |
|----------|---------|---------|
| `NEMOCLAW_OLLAMA_PULL_TIMEOUT` | `1800` (30 minutes) | Wall-clock timeout for `ollama pull` during onboard, in seconds. Accepts integer or float values. Already-downloaded layers are kept; re-running the pull resumes them. |
| `NEMOCLAW_LOCAL_INFERENCE_TIMEOUT` | `180` | Wall-clock timeout for the inference-server validation probe during onboard, in seconds. Raise on slow networks or for very large prompts. |
| `NEMOCLAW_SANDBOX_READY_TIMEOUT` | `180` | Wall-clock timeout for the post-create readiness wait, in seconds. Raise when the sandbox image build, gateway upload, or in-sandbox boot exceeds the default (typical on 70B+ models, first-time gateway uploads over slow links, or DGX Station / remote-VM first runs). When the deadline expires onboarding deletes the orphaned sandbox and prints the retry hint. |
| `NEMOCLAW_SANDBOX_READY_TIMEOUT` | `180` | Wall-clock timeout for post-create readiness and OpenShell command re-registration after policy application, in seconds. Raise when the sandbox image build, gateway upload, in-sandbox boot, or post-policy re-registration exceeds the default (typical on 70B+ models, first-time gateway uploads over slow links, or DGX Station / remote-VM first runs). When the post-create deadline expires, onboarding deletes an orphaned sandbox and prints the retry hint. |
| `NEMOCLAW_SANDBOX_READY_ERROR_DEBOUNCE` | `30` | Consecutive `Error`-phase polls (2s apart, so ~60s by default) the post-create readiness wait tolerates before treating `Error` as terminal. The gateway can briefly report a just-created sandbox in `Error` while it re-registers the sandbox (seen on DGX Spark); the debounce lets that transient recover to `Ready`. `Failed` and `CrashLoopBackOff` always fail immediately. Set to `1` to restore fast-fail on the first `Error` poll. |

<AgentOnly variant="openclaw,hermes">

For managed recovery, the same timeout covers OpenShell re-registration after transactional recreation.
When the deadline expires, the primary dashboard or API host forward stays stopped.

</AgentOnly>

```bash
export NEMOCLAW_OLLAMA_PULL_TIMEOUT=3600
export NEMOCLAW_SANDBOX_READY_TIMEOUT=600
$$nemoclaw onboard
```

If a timeout fires, onboarding emits the elapsed budget plus a hint to raise the relevant variable.
If the Ollama pull or post-create readiness timeout fires, onboarding emits the elapsed budget plus a hint to raise the relevant variable.
The Ollama pull preserves its partial download for the next attempt.
The readiness wait deletes the orphaned sandbox first so the next `$$nemoclaw onboard` starts clean.
The post-create readiness wait deletes the orphaned sandbox first so the next `$$nemoclaw onboard` starts clean.
A post-policy re-registration failure leaves the sandbox in place and reports that OpenShell did not re-register it.

### Lifecycle Behavior Flags

Expand Down
3 changes: 3 additions & 0 deletions docs/reference/troubleshooting.mdx
Original file line number Diff line number Diff line change
Expand Up @@ -1485,6 +1485,9 @@ openshell sandbox list
$$nemoclaw <name> status
```

If onboarding instead reports that the sandbox "did not re-register with OpenShell after policy application," the same timeout controls that post-policy command-readiness probe.
Raise the budget before retrying, then inspect the same gateway and sandbox status if re-registration still fails.

### Sandbox onboard fails with "entered Error phase before it became ready"

Onboarding ends with:
Expand Down
95 changes: 94 additions & 1 deletion src/lib/actions/sandbox/forward-recovery.ts
Original file line number Diff line number Diff line change
@@ -1,8 +1,14 @@
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
// SPDX-License-Identifier: Apache-2.0

import { spawnSync } from "node:child_process";

import { resolveOpenshell } from "../../adapters/openshell/resolve";
import { captureOpenshell, isCommandTimeout, runOpenshell } from "../../adapters/openshell/runtime";
import { OPENSHELL_PROBE_TIMEOUT_MS } from "../../adapters/openshell/timeouts";
import {
OPENSHELL_OPERATION_TIMEOUT_MS,
OPENSHELL_PROBE_TIMEOUT_MS,
} from "../../adapters/openshell/timeouts";
import * as agentRuntime from "../../agent/runtime";
import { DASHBOARD_PORT } from "../../core/ports";
import { waitUntil } from "../../core/wait";
Expand All @@ -11,9 +17,12 @@ import type { SandboxMessagingHostForwardPlan } from "../../messaging/manifest";
import { hydrateDerivedSandboxMessagingPlanFields } from "../../messaging/persistence";
import { parseSandboxMessagingPlan } from "../../messaging/plan-validation";
import { isRemoteDashboardBindRequested } from "../../onboard/dockerfile-remote-dashboard-bind-contract";
import { resolveSandboxGatewayName } from "../../onboard/gateway-binding";
import { isWsl } from "../../platform";
import { ROOT } from "../../state/paths";
import * as registry from "../../state/registry";
import { parseForwardList } from "../../state/sandbox-session";
import { buildSubprocessEnv } from "../../subprocess-env";
import {
classifyForwardHealthWithReachability,
isLocalForwardReachable,
Expand All @@ -38,10 +47,52 @@ type SandboxForwardRecoveryOptions = {
isWsl?: boolean;
};

type DashboardForwardStopRunner = (
args: string[],
options: { ignoreError: true; stdio: "ignore"; timeout: number },
) => { status?: number | null };

const FORWARD_RELEASE_TIMEOUT_MS = 5_000;
const FORWARD_RELEASE_POLL_MS = 250;

function isValidPort(value: unknown): value is number {
return typeof value === "number" && Number.isInteger(value) && value >= 1 && value <= 65535;
}

function runDashboardForwardStopBestEffort(
args: string[],
options: { timeout: number },
): { status?: number | null } {
try {
const openshellBinary = resolveOpenshell();
if (!openshellBinary) return { status: 1 };
return spawnSync(openshellBinary, args, {
cwd: ROOT,
env: buildSubprocessEnv(),
stdio: "ignore",
timeout: options.timeout,
});
} catch {
// The container lifecycle action has already completed; cleanup must not
// replace that result when OpenShell cannot be launched.
return { status: 1 };
}
}

function confirmDashboardForwardReleased(
port: number,
isForwardReachable: (port: number) => boolean,
): boolean {
const now = Date.now;
return waitUntil(() => !isForwardReachable(port), {
deadlineMs: now() + FORWARD_RELEASE_TIMEOUT_MS,
initialIntervalMs: FORWARD_RELEASE_POLL_MS,
maxIntervalMs: FORWARD_RELEASE_POLL_MS,
backoffFactor: 1,
now,
});
}

export function resolveSandboxDashboardPort(
sandboxName: string,
deps: SandboxPortDeps = {},
Expand All @@ -61,6 +112,48 @@ export function resolveSandboxDashboardPort(
return DASHBOARD_PORT;
}

/**
* Tear down the host-side dashboard port-forward this sandbox created.
*
* `stop` stops the container but must also release the forward it spawned;
* leaving it alive orphans an `ssh -L` listener on the dashboard port, which
* `status` then misreports as a foreign `sandbox_dashboard_port_conflict` and
* which `start`/`recover` contend with (#7227). Best-effort: a stop must still
* free container resources when openshell is unreachable, so errors are ignored
* — mirroring the sandbox- and gateway-scoped forward cleanup used elsewhere.
* OpenShell may return before its SSH listener exits, so successful commands
* also receive a bounded host-port release wait.
*/
export function teardownSandboxDashboardForward(
sandboxName: string,
deps: {
getSandbox?: typeof registry.getSandbox;
isLocalForwardReachable?: typeof isLocalForwardReachable;
resolveSandboxDashboardPort?: typeof resolveSandboxDashboardPort;
resolveSandboxGatewayName?: typeof resolveSandboxGatewayName;
runOpenshell?: DashboardForwardStopRunner;
} = {},
): void {
try {
const getSandbox = deps.getSandbox ?? registry.getSandbox;
const sandbox = getSandbox(sandboxName);
if (!sandbox) return;
const gatewayName = (deps.resolveSandboxGatewayName ?? resolveSandboxGatewayName)(sandbox);
const resolvePort = deps.resolveSandboxDashboardPort ?? resolveSandboxDashboardPort;
const port = resolvePort(sandboxName, { getSandbox: () => sandbox });
const run = deps.runOpenshell ?? runDashboardForwardStopBestEffort;
const result = run(["forward", "stop", String(port), sandboxName, "--gateway", gatewayName], {
ignoreError: true,
stdio: "ignore",
timeout: OPENSHELL_OPERATION_TIMEOUT_MS,
});
if (result.status !== 0) return;
confirmDashboardForwardReleased(port, deps.isLocalForwardReachable ?? isLocalForwardReachable);
} catch {
// Defense in depth for injected or future runners: teardown is best-effort.
}
}

/**
* Re-establish the dashboard port forward to the sandbox.
* Uses the recorded dashboard port when available, including custom ports for
Expand Down
Loading
Loading