fix(kubernetes): resume dormant managed hosts (#5046)

Signed-off-by: dbczumar <corey.zumar@databricks.com>
This commit is contained in:
Corey Zumar
2026-08-20 09:50:55 -07:00
committed by GitHub
parent ccaa42e926
commit 0b0fc3f2f9
2 changed files with 41 additions and 3 deletions
+23 -3
View File
@@ -16,7 +16,7 @@ creates the Job — an init container prepares the workspace (``mkdir`` + option
which dials back over the existing managed launch-token tunnel. Because the host
is never started by ``exec``-ing into an already-running container, this launcher
needs no ``pods/exec`` rights and no exec transport — it implements only
``prepare`` / ``provision`` / ``start_host`` / ``terminate``.
``prepare`` / ``provision`` / ``start_host`` / ``resume`` / ``terminate``.
Platform notes that shape this launcher:
@@ -982,7 +982,9 @@ class KubernetesSandboxLauncher(SandboxHostLauncher):
Server-managed only and entrypoint-as-host: :meth:`provision` reserves a Job
name, :meth:`start_host` creates a per-Job token Secret and a Job whose Pod
template's init container prepares the workspace and whose main container runs
``omnigent host``, and :meth:`terminate` deletes both. The Job uses
``omnigent host``. :meth:`resume` removes a dormant Job and its stale token
Secret so the managed-host wake path can recreate both under the same sandbox
id, while :meth:`terminate` permanently deletes them. The Job uses
``restartPolicy: OnFailure`` so the kubelet automatically restarts a crashed
host container, providing automatic failover within the Job's
``backoffLimit``. All transport rides the official ``kubernetes`` client's
@@ -992,6 +994,7 @@ class KubernetesSandboxLauncher(SandboxHostLauncher):
"""
provider: ClassVar[str] = "kubernetes"
can_resume: ClassVar[bool] = True
@property
def capabilities(self) -> SandboxCapabilities:
@@ -999,7 +1002,7 @@ class KubernetesSandboxLauncher(SandboxHostLauncher):
cli_bootstrap=False,
managed_launch=True,
local_port_forward=False,
resume_stopped=False,
resume_stopped=True,
programmatic_terminate=True,
classifies_runner_by_agent=True,
)
@@ -1739,6 +1742,23 @@ class KubernetesSandboxLauncher(SandboxHostLauncher):
if first_error is not None:
raise first_error
def resume(self, sandbox_id: str) -> None:
"""
Prepare a dormant Kubernetes sandbox for recreation in place.
Kubernetes Jobs cannot be restarted after their host process exits.
Remove the old Job and launch-token Secret so the shared managed-host
wake path can call :meth:`start_host` with the same sandbox id and a
freshly armed token. Operator-managed PVCs are external resources and
are not touched.
:param sandbox_id: The dormant Job name to recreate.
:raises click.ClickException: On an API delete failure other than
not-found.
"""
click.echo(f"▸ Resuming Kubernetes sandbox '{sandbox_id}'")
self.terminate(sandbox_id)
def _delete_with_retry(self, kind: str, name: str, delete: Callable[[], object]) -> None:
"""
Run *delete* with bounded retries on a transient timeout/connection
@@ -932,6 +932,24 @@ def test_terminate_deletes_job_and_secret(
assert core.deleted_secrets == ["omnigent-job-6-token"]
def test_resume_recycles_job_and_token_secret(
fake_clients: tuple[_FakeCore, _FakeBatch],
) -> None:
"""Resume clears stale launch resources so start_host can recreate the same id."""
core, batch = fake_clients
launcher = _launcher()
assert launcher.can_resume is True
assert launcher.capabilities.resume_stopped is True
launcher.resume("omnigent-job-resume")
assert batch.deleted_jobs == ["omnigent-job-resume"]
assert batch.last_delete_body.propagation_policy == "Foreground"
assert core.deleted_pods == ["omnigent-job-resume"]
assert core.deleted_secrets == ["omnigent-job-resume-token"]
def test_terminate_is_idempotent_on_404(
fake_clients: tuple[_FakeCore, _FakeBatch],
) -> None: