From 3434e4c26e2c2fca5f0328c5afabd85399cf5db1 Mon Sep 17 00:00:00 2001 From: Eric Charles Date: Sat, 26 Sep 2026 10:14:01 +0200 Subject: [PATCH 1/6] Marimo reactivity through the Jupyter-shaped API: a named cell via the context, reactions typed on the result, in the reply, in the hook messages and the stream, the graph through the client; 1.9.39 --- CHANGELOG.md | 16 ++++ code_sandboxes/__version__.py | 2 +- code_sandboxes/client.py | 121 +++++++++++++++++++++++++++---- code_sandboxes/marimo_sandbox.py | 86 ++++++++++++++++++---- code_sandboxes/models.py | 28 +++++++ docs/docs/providers/marimo.mdx | 48 +++++++++++- tests/test_marimo_sandbox.py | 91 ++++++++++++++++++++++- 7 files changed, 357 insertions(+), 35 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7045ef6..2568e99 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,22 @@ ## Unreleased +## 1.9.39 + +- **Marimo reactivity through the Jupyter-shaped API** (#37). `run_code` + names its cell through the execution context (`Context(id=...)`; the same + id replaces the cell) and the result says what happened: `cell_id`, and + `reactions` — every cell re-run because of it, each a `Reaction` with its + code and its own `ExecutionResult`. `CodeSandboxClient.execute`, + `execute_code` and `execute_code_streaming` take `cell_id`; a reply carries + the reactions under `marimo` with their own Jupyter-shaped outputs; + `execute_interactive` emits them after the cell's own, each message tagged + `metadata.marimo = {cell_id, reaction}`; the stream tags every event with + `marimo_cell_id`. The client passes `run_cell`, `register_cell`, + `remove_cell`, `plan`, `graph` and `cells` through to a reactive sandbox and + says so with `reactive`. The `marimo_reactions` extra attribute is replaced + by the typed `reactions` field. Nothing changes on the wire. + ## 1.9.38 - **The environments line and main are one branch again**: everything diff --git a/code_sandboxes/__version__.py b/code_sandboxes/__version__.py index e4f4108..5e178af 100644 --- a/code_sandboxes/__version__.py +++ b/code_sandboxes/__version__.py @@ -3,4 +3,4 @@ """Code Sandboxes.""" -__version__ = "1.9.38" +__version__ = "1.9.39" diff --git a/code_sandboxes/client.py b/code_sandboxes/client.py index 4ac2604..4690d51 100644 --- a/code_sandboxes/client.py +++ b/code_sandboxes/client.py @@ -60,6 +60,7 @@ from .filesystem import FileInfo, SandboxFilesystem from .models import ( CodeError, + Context, ExecutionResult, OutputMessage, Result, @@ -114,11 +115,29 @@ def execution_result_to_reply(execution: ExecutionResult) -> dict[str, Any]: } ) - return { + reply: dict[str, Any] = { "execution_count": execution.execution_count, "outputs": outputs, "status": "ok" if execution.success else "error", } + if execution.cell_id is not None or execution.reactions: + # A reactive sandbox (Marimo): which cell this was, and what else + # ran because of it, each with its own Jupyter-shaped outputs — so a + # caller that only reads replies still sees every changed cell. + reply["marimo"] = { + "cell_id": execution.cell_id, + "reactions": [ + { + "cell_id": reaction.cell_id, + "code": reaction.code, + "status": "error" if reaction.result.code_error is not None else "ok", + "execution_count": reaction.result.execution_count, + "outputs": execution_result_to_reply(reaction.result)["outputs"], + } + for reaction in execution.reactions + ], + } + return reply @dataclass @@ -405,19 +424,33 @@ async def close_async(self) -> None: else: self.close() + def _cell_kwargs(self, cell_id: str | None) -> dict[str, Any]: + """The execution context that names a cell, only when one was named. + + Passed as a keyword only then: a sandbox with no notion of cells — + every variant but Marimo, and any subclass someone wrote against the + older signatures — is called exactly as before. + """ + return {"context": Context(id=cell_id)} if cell_id else {} + def execute_code( self, code: str, language: str = "python", timeout: float | None = None, envs: dict[str, str] | None = None, + cell_id: str | None = None, ) -> CodeExecutionOutcome: """Execute code and return a normalized outcome. - The sandbox is started automatically if needed. + The sandbox is started automatically if needed. ``cell_id`` names the + cell on a reactive sandbox (Marimo): the same id again replaces the + cell, and the cells depending on it re-run. """ self.start() - execution = self._sandbox.run_code(code, language=language, timeout=timeout, envs=envs) + execution = self._sandbox.run_code( + code, language=language, timeout=timeout, envs=envs, **self._cell_kwargs(cell_id) + ) return CodeExecutionOutcome.from_execution_result(execution) def execute( @@ -425,12 +458,17 @@ def execute( code: str, silent: bool = False, timeout: float | None = None, + cell_id: str | None = None, **kwargs: Any, ) -> dict[str, Any]: - """Execute code and return a backend-neutral Jupyter-shaped reply.""" + """Execute code and return a backend-neutral Jupyter-shaped reply. + + On a reactive sandbox the reply also carries ``marimo``: the cell this + ran as and the reactions, each with its own outputs. + """ del silent, kwargs self.start() - execution = self._sandbox.run_code(code, timeout=timeout) + execution = self._sandbox.run_code(code, timeout=timeout, **self._cell_kwargs(cell_id)) return execution_result_to_reply(execution) def execute_interactive( @@ -457,21 +495,75 @@ def execute_interactive( """ reply = self.execute(code, silent=silent, timeout=timeout, **kwargs) if output_hook is not None: - for output in reply["outputs"]: - msg_type = output.get("output_type", "display_data") - output_hook( - { + marimo = reply.get("marimo") or {} + # The cell's own outputs, then each reaction's: every message + # says which cell it belongs to, the way a kernel's IOPub + # messages carry their parent header. + batches = [(marimo.get("cell_id"), False, reply["outputs"])] + [ + (reaction["cell_id"], True, reaction["outputs"]) + for reaction in marimo.get("reactions", []) + ] + for cell_id, is_reaction, outputs in batches: + for output in outputs: + msg_type = output.get("output_type", "display_data") + message: dict[str, Any] = { "header": {"msg_type": msg_type}, "msg_type": msg_type, "content": output, } - ) + if cell_id is not None: + message["metadata"] = { + "marimo": {"cell_id": cell_id, "reaction": is_reaction} + } + output_hook(message) reply["content"] = { "status": reply.get("status", "ok"), "execution_count": reply.get("execution_count"), } return reply + # -- reactivity (the Marimo variant) ------------------------------------ + + @property + def reactive(self) -> bool: + """Whether the sandbox keeps a cell graph and re-runs dependents.""" + return hasattr(self._sandbox, "run_cell") + + def _reactive_sandbox(self) -> Any: + if not self.reactive: + raise TypeError( + f"The {self.variant or 'current'} sandbox is not reactive; " + "cells and reactions are the marimo variant's." + ) + self.start() + return self._sandbox + + def run_cell( + self, cell_id: str, code: str, *, react: bool = True, timeout: float | None = None + ) -> Any: + """Run one named cell, then the cells that depend on it (`MarimoRun`).""" + return self._reactive_sandbox().run_cell(cell_id, code, react=react, timeout=timeout) + + def register_cell(self, cell_id: str, code: str) -> dict[str, Any]: + """Put a cell in the graph without running it; answers its names.""" + return self._reactive_sandbox().register_cell(cell_id, code) + + def remove_cell(self, cell_id: str) -> None: + self._reactive_sandbox().remove_cell(cell_id) + + def plan(self, cell_id: str) -> list[str]: + """The cells that re-run after `cell_id`, in dependency order.""" + return self._reactive_sandbox().plan(cell_id) + + def graph(self) -> dict[str, Any]: + """Every cell's names and neighbours, the conflicts and the cycles.""" + return self._reactive_sandbox().graph() + + @property + def cells(self) -> dict[str, str]: + """The registered cells' source, by id.""" + return dict(self._reactive_sandbox().cells) + def get_variable(self, name: str) -> Any: """Read a variable through the wrapped sandbox.""" self.start() @@ -546,18 +638,17 @@ def execute_code_streaming( language: str = "python", timeout: float | None = None, envs: dict[str, str] | None = None, + cell_id: str | None = None, ) -> Iterator[StreamingItem]: """Execute code and stream output events. This is a thin variant-agnostic wrapper over - ``Sandbox.run_code_streaming``. + ``Sandbox.run_code_streaming``. On a reactive sandbox the reactions' + events follow the cell's own, each tagged with ``marimo_cell_id``. """ self.start() yield from self._sandbox.run_code_streaming( - code, - language=language, - timeout=timeout, - envs=envs, + code, language=language, timeout=timeout, envs=envs, **self._cell_kwargs(cell_id) ) async def execute_code_streaming_async( diff --git a/code_sandboxes/marimo_sandbox.py b/code_sandboxes/marimo_sandbox.py index 35321be..ced2811 100644 --- a/code_sandboxes/marimo_sandbox.py +++ b/code_sandboxes/marimo_sandbox.py @@ -4,6 +4,16 @@ """A Marimo sandbox: a Jupyter kernel with Marimo's reactivity in it. +Reactivity crosses the Jupyter-shaped API too (code-sandboxes#37). A caller +that only knows `run_code` / `CodeSandboxClient.execute` names the cell it is +running through the execution context (`Context(id="cell-a")`) and gets, on +the result, the cell it ran as (`cell_id`) and every cell re-run because of +it (`reactions`, each with its own result). `run_code_streaming` yields the +reactions' events after the cell's own, each event tagged with the cell it +came from (`marimo_cell_id`), so a stream consumer can route outputs to the +right cell. Nothing on the wire changes: the graph is still driven through +ordinary execute requests and answers on stdout. + Marimo notebooks are reactive — running a cell re-runs every cell that depends on what it defined — and Marimo works that out from the cells' source, not from running them (code-sandboxes#34). So a Marimo sandbox is a Jupyter server @@ -26,6 +36,7 @@ import logging import time +from collections.abc import Iterator from dataclasses import dataclass, field from typing import Any @@ -43,6 +54,7 @@ ExecutionResult, OutputHandler, OutputMessage, + Reaction, Result, SandboxEnvironment, ) @@ -231,6 +243,7 @@ def run_cell( registration=registration, ) result = self._plain_run(code, timeout=timeout, **handlers) + result.cell_id = cell_id run = MarimoRun(cell_id=cell_id, result=result, registration=registration) if not react or not run.ok: return run @@ -243,9 +256,15 @@ def run_cell( if self._interrupt_requested.is_set(): break dependent_result = self._plain_run(dependent_code, timeout=timeout, **handlers) + dependent_result.cell_id = dependent run.reactions.append( CellRun(cell_id=dependent, code=dependent_code, result=dependent_result) ) + # The result the caller holds says what else ran: a Jupyter + # protocol consumer learns the reactions from it. + result.reactions.append( + Reaction(cell_id=dependent, code=dependent_code, result=dependent_result) + ) if dependent_result.code_error is not None: break return run @@ -263,10 +282,12 @@ def run_code( # type: ignore[override] envs: dict[str, str] | None = None, timeout: float | None = None, ) -> ExecutionResult: - """Run code as a cell of its own, reactively; answers that cell's result. + """Run code as a cell, reactively; answers that cell's result. - The cells it made re-run are on the result's ``marimo_reactions`` - (cell ids), so a caller that wants them can ask `cells` for their code. + The cell is the context's id when the caller gave one — the same id + again replaces the cell rather than adding a second — and a cell of + its own otherwise. The cells it made re-run are on the result's + ``reactions``, each with its code and its own result. The execution window is open for the whole run; each cell inside it is a parent `run_code`, which closes the window on its way out and @@ -279,9 +300,14 @@ def run_code( # type: ignore[override] if envs: env_code = "\n".join(f"import os; os.environ[{k!r}] = {v!r}" for k, v in envs.items()) self._plain_run(env_code) - self._anonymous += 1 + cell_id = ( + context.id if context is not None and context.id and context.id != "default" else None + ) + if cell_id is None: + self._anonymous += 1 + cell_id = f"cell-{self._anonymous}" run = self.run_cell( - f"cell-{self._anonymous}", + cell_id, code, on_stdout=on_stdout, on_stderr=on_stderr, @@ -289,15 +315,47 @@ def run_code( # type: ignore[override] on_error=on_error, timeout=timeout, ) - result = run.result - if run.reactions: - # `ExecutionResult` allows extra fields; the ids are enough to - # find the cells again, and their results are the reactions'. - try: - result.marimo_reactions = [reaction.cell_id for reaction in run.reactions] # type: ignore[attr-defined] - except (AttributeError, ValueError): - pass - return result + return run.result + + def run_code_streaming( # type: ignore[override] + self, + code: str, + language: str = "python", + context: Context | None = None, + envs: dict[str, str] | None = None, + timeout: float | None = None, + ) -> Iterator[OutputMessage | Result | CodeError]: + """Stream the cell's events, then each reaction's, every event tagged. + + Every yielded item carries ``marimo_cell_id`` (the cell it came from) + and ``marimo_reaction`` (False for the cell that was run, True for a + cell re-run because of it), so a consumer that reads the stream as a + Jupyter client does can still tell the cells apart. + """ + execution = self.run_code( + code, language=language, context=context, envs=envs, timeout=timeout + ) + for cell_id, reaction, result in [ + (execution.cell_id, False, execution), + *((r.cell_id, True, r.result) for r in execution.reactions), + ]: + for item in self._events_of(result): + item.marimo_cell_id = cell_id # type: ignore[attr-defined] + item.marimo_reaction = reaction # type: ignore[attr-defined] + yield item + + @staticmethod + def _events_of(execution: ExecutionResult) -> Iterator[OutputMessage | Result | CodeError]: + """One result's events, in the order the base streaming yields them.""" + yield from execution.logs.stdout + yield from execution.logs.stderr + yield from execution.results + if not execution.execution_ok and execution.execution_error: + yield CodeError( + name="SandboxExecutionError", value=execution.execution_error, traceback="" + ) + if execution.code_error: + yield execution.code_error def __repr__(self) -> str: return f"MarimoSandbox(cells={len(self._cells)}, helper={HELPER_NAME!r})" diff --git a/code_sandboxes/models.py b/code_sandboxes/models.py index d4abdb3..384b823 100644 --- a/code_sandboxes/models.py +++ b/code_sandboxes/models.py @@ -397,6 +397,22 @@ def __repr__(self) -> str: return f"Context(id={self.id!r}, language={self.language!r})" +class Reaction(BaseModel): + """One cell re-run because the cell that just ran changed what it reads. + + A reactive sandbox (the Marimo variant) runs, after every execution, the + cells that depend on it, in dependency order. Each of those runs is a + reaction: the cell, its code, and the result it produced, so a caller who + only speaks the Jupyter protocol still learns what else changed. + """ + + model_config = ConfigDict(extra="allow") + + cell_id: str + code: str + result: "ExecutionResult" + + class ExecutionResult(BaseModel): """Complete result of a code execution. @@ -471,6 +487,15 @@ class ExecutionResult(BaseModel): "None means no explicit exit." ), ) + # Reactivity (the Marimo variant; empty everywhere else) + cell_id: Optional[str] = Field( + default=None, + description="The cell this execution ran as, on a sandbox that keeps a cell graph", + ) + reactions: list[Reaction] = Field( + default_factory=list, + description="The cells re-run because they depend on this one, in the order they ran", + ) @property def text(self) -> Optional[str]: @@ -706,3 +731,6 @@ class JupyterServerOptions(BaseModel): token: Optional[str] = Field(default=None, repr=False) install_if_missing: bool = True install_timeout: float = 180.0 + + +Reaction.model_rebuild() diff --git a/docs/docs/providers/marimo.mdx b/docs/docs/providers/marimo.mdx index 9fd1c29..d13fc63 100644 --- a/docs/docs/providers/marimo.mdx +++ b/docs/docs/providers/marimo.mdx @@ -45,10 +45,50 @@ running it, `remove_cell` takes it out, `plan(cell_id)` lists what would re-run, `graph()` snapshots every cell's names, neighbours, conflicts and cycles, and `cells` is the registered source by id. -`run_code` keeps working as on every sandbox — each call is a cell of its own, -so an agent that never heard of cells still gets a consistent state: what it -ran earlier and that depends on what it just ran is run again. The ids of the -cells that re-ran are on the result as `marimo_reactions`. +## Through the Jupyter-shaped API + +`run_code` keeps working as on every sandbox, and it is reactive too. Name the +cell through the execution context — `Context(id="a")` — and the same id again +replaces that cell rather than adding a second; leave it out and each call is a +cell of its own, so an agent that never heard of cells still gets a consistent +state: what it ran earlier and that depends on what it just ran is run again. + +The result says what happened: `cell_id` is the cell it ran as, and +`reactions` lists every cell re-run because of it, in the order they ran, each +a `Reaction` with the cell's `code` and its own `ExecutionResult`. A reaction +that fails stops the chain and is reported, not hidden. + +`CodeSandboxClient` carries the same through the Jupyter protocol, so a +consumer that only reads replies and streams loses nothing: + +```python +from code_sandboxes import CodeSandboxClient + +client = CodeSandboxClient.create(variant="marimo") +client.execute("n = 1", cell_id="a") +client.execute("print(n * 3)", cell_id="b") +reply = client.execute("n = 4", cell_id="a") +reply["marimo"]["cell_id"] # "a" +reply["marimo"]["reactions"][0]["cell_id"] # "b" +reply["marimo"]["reactions"][0]["outputs"] # [{"output_type": "stream", ..., "text": "12\n"}] +``` + +- `execute`, `execute_code` and `execute_code_streaming` take `cell_id`; a + reply with reactions carries them under `marimo`, each with its own + Jupyter-shaped `outputs`, `status` and `execution_count`. +- `execute_interactive` emits the reactions' outputs after the cell's own, + every message tagged `metadata.marimo = {cell_id, reaction}` — the way a + kernel's IOPub messages carry their parent — so a client routes each output + to the cell it belongs to. +- the stream from `execute_code_streaming` yields the reactions' events after + the cell's, each tagged `marimo_cell_id` and `marimo_reaction`. +- `client.reactive` says whether the sandbox keeps a graph, and `run_cell`, + `register_cell`, `remove_cell`, `plan`, `graph` and `cells` pass through to + it; on any other variant they raise `TypeError`. + +Nothing changes on the wire: the graph is driven through ordinary execute +requests and answers on stdout, which is what keeps `jupyter-kernel-client` +the client API. The concrete implementation is available from a top-level module: diff --git a/tests/test_marimo_sandbox.py b/tests/test_marimo_sandbox.py index 1e1bb44..d2fec8e 100644 --- a/tests/test_marimo_sandbox.py +++ b/tests/test_marimo_sandbox.py @@ -179,7 +179,10 @@ def test_run_code_is_a_cell_of_its_own_and_reacts(sandbox): sandbox.run_code("print(base * 2)") result = sandbox.run_code("base = 7") assert result.code_error is None - assert getattr(result, "marimo_reactions", None) == ["cell-2"] + assert result.cell_id == "cell-3" + assert [reaction.cell_id for reaction in result.reactions] == ["cell-2"] + assert result.reactions[0].code == "print(base * 2)" + assert result.reactions[0].result.logs.stdout[-1].line == "14" assert sandbox.kernel_client.executed[-1] == "print(base * 2)" @@ -207,3 +210,89 @@ def test_the_variant_is_known(): assert normalize_variant(SandboxVariant.MARIMO) == "marimo" assert [env.name for env in Sandbox.list_environments("marimo")] == ["marimo"] + + +# -- the Jupyter-shaped API carries the reactivity (code-sandboxes#37) -------- + + +def test_a_context_id_names_the_cell_and_the_same_id_replaces_it(sandbox): + from code_sandboxes.models import Context + + first = sandbox.run_code("n = 1", context=Context(id="a")) + sandbox.run_code("print(n + 1)", context=Context(id="b")) + again = sandbox.run_code("n = 5", context=Context(id="a")) + assert first.cell_id == "a" and again.cell_id == "a" + assert sorted(sandbox.cells) == ["a", "b"] + assert [reaction.cell_id for reaction in again.reactions] == ["b"] + assert again.reactions[0].result.logs.stdout[-1].line == "6" + + +def test_the_jupyter_shaped_reply_carries_the_reactions(sandbox): + from code_sandboxes import CodeSandboxClient + + client = CodeSandboxClient(sandbox) + client.execute("n = 1", cell_id="a") + client.execute("print(n * 3)", cell_id="b") + reply = client.execute("n = 4", cell_id="a") + assert reply["status"] == "ok" + assert reply["marimo"]["cell_id"] == "a" + (reaction,) = reply["marimo"]["reactions"] + assert reaction["cell_id"] == "b" and reaction["status"] == "ok" + assert reaction["outputs"] == [{"output_type": "stream", "name": "stdout", "text": "12\n"}] + + +def test_execute_interactive_tags_every_message_with_its_cell(sandbox): + from code_sandboxes import CodeSandboxClient + + client = CodeSandboxClient(sandbox) + client.execute("n = 1", cell_id="a") + client.execute("print('b sees', n)", cell_id="b") + seen = [] + client.execute_interactive("n = 2\nprint('a ran')", cell_id="a", output_hook=seen.append) + tags = [(m["metadata"]["marimo"]["cell_id"], m["metadata"]["marimo"]["reaction"]) for m in seen] + assert tags == [("a", False), ("b", True)] + assert seen[1]["content"]["text"] == "b sees 2\n" + + +def test_streaming_yields_the_reactions_events_tagged(sandbox): + from code_sandboxes import CodeSandboxClient + + client = CodeSandboxClient(sandbox) + client.execute("n = 1", cell_id="a") + client.execute("print(n)", cell_id="b") + events = list(client.execute_code_streaming("n = 9", cell_id="a")) + assert [(e.marimo_cell_id, e.marimo_reaction) for e in events] == [("b", True)] + assert events[0].line == "9" + + +def test_a_failing_reaction_is_reported_not_hidden(sandbox): + from code_sandboxes import CodeSandboxClient + + client = CodeSandboxClient(sandbox) + client.execute("n = 1", cell_id="a") + client.execute("assert n < 5", cell_id="b") + client.execute("print('after b')", cell_id="c") # depends on nothing: never re-run + reply = client.execute("n = 10", cell_id="a") + assert reply["status"] == "ok" # the cell itself ran + (reaction,) = reply["marimo"]["reactions"] + assert reaction["cell_id"] == "b" and reaction["status"] == "error" + assert reaction["outputs"][0]["output_type"] == "error" + + +def test_the_client_offers_the_graph_to_a_reactive_sandbox_only(sandbox): + from code_sandboxes import CodeSandboxClient + from code_sandboxes.eval_sandbox import EvalSandbox + + client = CodeSandboxClient(sandbox) + assert client.reactive is True + client.run_cell("a", "x = 1") + client.run_cell("b", "y = x") + assert client.plan("a") == ["b"] + assert set(client.graph()["cells"]) == {"a", "b"} + client.remove_cell("b") + assert client.cells == {"a": "x = 1"} + + plain = CodeSandboxClient(EvalSandbox()) + assert plain.reactive is False + with pytest.raises(TypeError, match="not reactive"): + plain.plan("a") From c39f6d29894fbb25f30e9a153a5d4ac5609fe05a Mon Sep 17 00:00:00 2001 From: Eric Charles Date: Sat, 26 Sep 2026 10:26:25 +0200 Subject: [PATCH 2/6] Review: a reaction's status and the chain's stop use the whole success predicate, a refused cell is still named, Reaction is exported; MarimoCells drives the same reactivity over any Jupyter-shaped client --- code_sandboxes/__init__.py | 6 + code_sandboxes/client.py | 5 +- code_sandboxes/marimo_cells.py | 261 ++++++++++++++++++++++++++++++ code_sandboxes/marimo_reactive.py | 5 + code_sandboxes/marimo_sandbox.py | 8 +- tests/test_marimo_cells.py | 140 ++++++++++++++++ tests/test_marimo_sandbox.py | 19 +++ 7 files changed, 442 insertions(+), 2 deletions(-) create mode 100644 code_sandboxes/marimo_cells.py create mode 100644 tests/test_marimo_cells.py diff --git a/code_sandboxes/__init__.py b/code_sandboxes/__init__.py index 6ed6384..a0093f6 100644 --- a/code_sandboxes/__init__.py +++ b/code_sandboxes/__init__.py @@ -157,6 +157,7 @@ get_manager, manageable_variants, ) +from .marimo_cells import CellReply, CellsRun, MarimoCells from .marimo_sandbox import CellRun, MarimoRun, MarimoSandbox from .modal_sandbox import ModalSandbox from .models import ( @@ -170,6 +171,7 @@ MIMEType, OutputHandler, OutputMessage, + Reaction, ResourceConfig, Result, SandboxConfig, @@ -204,7 +206,9 @@ "RUNTIMES_API_PREFIX", "BuildEntry", "BuiltArtifact", + "CellReply", "CellRun", + "CellsRun", "CloudflareSandbox", "CodeError", "CodeExecutionOutcome", @@ -243,6 +247,7 @@ "Logs", "MIMEType", "ManifestLocation", + "MarimoCells", "MarimoRun", "MarimoSandbox", "MaterializeEntry", @@ -253,6 +258,7 @@ "PreparedAttachment", "ProcessHandle", "ProviderRequirement", + "Reaction", "ResourceConfig", "Result", "Sandbox", diff --git a/code_sandboxes/client.py b/code_sandboxes/client.py index 4690d51..3d0ab2e 100644 --- a/code_sandboxes/client.py +++ b/code_sandboxes/client.py @@ -130,7 +130,10 @@ def execution_result_to_reply(execution: ExecutionResult) -> dict[str, Any]: { "cell_id": reaction.cell_id, "code": reaction.code, - "status": "error" if reaction.result.code_error is not None else "ok", + # The same predicate as the reply's own status: a reaction + # that was interrupted, failed in the infrastructure or + # exited non-zero is not "ok" either. + "status": "ok" if reaction.result.success else "error", "execution_count": reaction.result.execution_count, "outputs": execution_result_to_reply(reaction.result)["outputs"], } diff --git a/code_sandboxes/marimo_cells.py b/code_sandboxes/marimo_cells.py new file mode 100644 index 0000000..a74c515 --- /dev/null +++ b/code_sandboxes/marimo_cells.py @@ -0,0 +1,261 @@ +# Copyright (c) 2025-2026 Datalayer, Inc. +# +# BSD 3-Clause License + +"""Marimo's reactive cells over any sandbox that speaks the Jupyter protocol. + +`MarimoSandbox` is one variant: a Jupyter *server* sandbox with the graph in +its kernel. A sandbox somewhere else — a Datalayer runtime, a Kaggle kernel, +anything a `CodeSandboxClient` wraps — is a kernel too, and the graph needs +nothing from a kernel but execute requests and stdout (`marimo_reactive`). So +this is the same reactivity as a *driver* over a client rather than as a +variant: install the helper into whatever kernel the client reaches, then +run cells through the client's Jupyter-shaped `execute`, re-running the +dependents the graph names. + + cells = MarimoCells(client) # any CodeSandboxClient + cells.run_cell("a", "x = 1") + run = cells.run_cell("b", "print(x)") + cells.run_cell("a", "x = 2").reactions[0].cell_id # "b", re-run + +This is what an MCP toolset uses to give an agent reactive cells on the +sandbox its session already holds (code-sandboxes#37). On a client whose +sandbox is a `MarimoSandbox`, the sandbox's own graph is used — the same cells +seen from either side. +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass, field +from typing import Any, Protocol + +from .marimo_reactive import ( + HELPER_NAME, + KERNEL_HELPER_SOURCE, + decode_answer, + question, +) + +logger = logging.getLogger(__name__) + +#: How the helper's answer is found in a reply: the stdout stream outputs. +_STDOUT = ("stream", "stdout") + + +class JupyterShaped(Protocol): + """What the driver needs of a client: `execute` answering a Jupyter-shaped reply.""" + + def execute(self, code: str, timeout: float | None = None, **kwargs: Any) -> dict[str, Any]: ... + + +@dataclass +class CellReply: + """One cell's execution, as the Jupyter-shaped reply the client answered.""" + + cell_id: str + code: str + reply: dict[str, Any] + + @property + def ok(self) -> bool: + return self.reply.get("status", "ok") == "ok" + + @property + def outputs(self) -> list[dict[str, Any]]: + return list(self.reply.get("outputs", [])) + + +@dataclass +class CellsRun: + """What running one cell did: the cell, then what it made re-run.""" + + cell: CellReply + reactions: list[CellReply] = field(default_factory=list) + #: The cell's names as the graph read them, or its `error` when the source + #: does not parse (the cell then never ran). + registration: dict[str, Any] = field(default_factory=dict) + + @property + def ok(self) -> bool: + return "error" not in self.registration and self.cell.ok + + def to_dict(self) -> dict[str, Any]: + """The run as plain data, the shape a tool answers with.""" + return { + "cell_id": self.cell.cell_id, + "status": "error" if not self.ok else "ok", + "outputs": self.cell.outputs, + "registration": self.registration, + "reactions": [ + { + "cell_id": reaction.cell_id, + "code": reaction.code, + "status": "ok" if reaction.ok else "error", + "outputs": reaction.outputs, + } + for reaction in self.reactions + ], + } + + +class MarimoCells: + """Marimo's reactive cell graph, driven through a client's `execute`. + + ``install_marimo`` says whether a kernel that lacks marimo gets it + installed with pip (once, at `install`); ``False`` refuses instead. + """ + + def __init__( + self, + client: JupyterShaped, + *, + install_marimo: bool = True, + timeout: float | None = None, + ): + self._client = client + self._install_marimo = install_marimo + self._timeout = timeout + self._ready = False + self._cells: dict[str, str] = {} + self._anonymous = 0 + + # -- the kernel ------------------------------------------------------ + + def _execute(self, code: str, timeout: float | None = None) -> dict[str, Any]: + return self._client.execute(code, timeout=timeout if timeout is not None else self._timeout) + + @staticmethod + def _stdout_lines(reply: dict[str, Any]) -> list[str]: + lines: list[str] = [] + for output in reply.get("outputs", []): + if (output.get("output_type"), output.get("name")) == _STDOUT: + lines.extend(str(output.get("text", "")).splitlines()) + return lines + + @staticmethod + def _error_of(reply: dict[str, Any]) -> str | None: + for output in reply.get("outputs", []): + if output.get("output_type") == "error": + return f"{output.get('ename', 'Error')}: {output.get('evalue', '')}" + return None if reply.get("status", "ok") == "ok" else "the execution failed" + + def install(self) -> None: + """Put the graph in the kernel, installing marimo first if it is missing. + + Idempotent: the helper keeps the graph it already holds, so a driver + created again on the same sandbox — another replica answering the + session, say — sees the cells registered before. + """ + if self._ready: + return + probe = self._execute("import marimo") + if self._error_of(probe) is not None: + if not self._install_marimo: + raise RuntimeError( + "The kernel has no marimo and install_marimo is False: " + "install marimo in the environment first." + ) + logger.info("Installing marimo into the kernel.") + installed = self._execute( + "import subprocess, sys\n" + "subprocess.check_call(" + "[sys.executable, '-m', 'pip', 'install', '--quiet', 'marimo'])" + ) + problem = self._error_of(installed) + if problem is not None: + raise RuntimeError(f"marimo could not be installed: {problem}") + bootstrap = self._execute(KERNEL_HELPER_SOURCE) + problem = self._error_of(bootstrap) + if problem is not None: + raise RuntimeError(f"The Marimo reactive helper could not start: {problem}") + self._ready = True + # Cells registered before this driver existed (the same kernel, another + # driver): read them back so `cells` and the reactions know their code. + for cell_id, code in self._ask("codes").items(): + self._cells.setdefault(cell_id, code) + + def _ask(self, method: str, *args: Any) -> Any: + self.install() + reply = self._execute(question(method, *args)) + problem = self._error_of(reply) + if problem is not None: + raise RuntimeError(f"The Marimo graph refused {method}: {problem}") + return decode_answer(self._stdout_lines(reply)) + + # -- the graph ------------------------------------------------------- + + def register_cell(self, cell_id: str, code: str) -> dict[str, Any]: + """Put a cell in the graph without running it; answers its names.""" + answer = self._ask("register", cell_id, code) + if "error" in answer: + self._cells.pop(cell_id, None) + else: + self._cells[cell_id] = code + return answer + + def remove_cell(self, cell_id: str) -> None: + self._cells.pop(cell_id, None) + self._ask("remove", cell_id) + + def plan(self, cell_id: str) -> list[str]: + """The cells that re-run after `cell_id`, in dependency order.""" + return [str(dep) for dep in self._ask("plan", cell_id)] + + def graph(self) -> dict[str, Any]: + """Every cell's names and neighbours, the conflicts and the cycles.""" + return self._ask("snapshot") + + @property + def cells(self) -> dict[str, str]: + """The registered cells' source, by id — the kernel's, read back first.""" + self.install() + return dict(self._cells) + + # -- running --------------------------------------------------------- + + def run_cell( + self, cell_id: str, code: str, *, react: bool = True, timeout: float | None = None + ) -> CellsRun: + """Run one cell, then the cells that depend on it. + + A cell that fails stops the reaction: nothing downstream runs on a + state the failure left half-made, which is what Marimo does too. + """ + registration = self.register_cell(cell_id, code) + if "error" in registration: + reply = { + "status": "error", + "outputs": [ + { + "output_type": "error", + "ename": "SyntaxError", + "evalue": registration["error"], + "traceback": [], + } + ], + } + return CellsRun(cell=CellReply(cell_id, code, reply), registration=registration) + run = CellsRun( + cell=CellReply(cell_id, code, self._execute(code, timeout)), + registration=registration, + ) + if not react or not run.ok: + return run + for dependent in self.plan(cell_id): + dependent_code = self._cells.get(dependent) + if dependent_code is None: + continue + reaction = CellReply(dependent, dependent_code, self._execute(dependent_code, timeout)) + run.reactions.append(reaction) + if not reaction.ok: + break + return run + + def run_code(self, code: str, *, timeout: float | None = None) -> CellsRun: + """Run code as a cell of its own, reactively.""" + self._anonymous += 1 + return self.run_cell(f"cell-{self._anonymous}", code, timeout=timeout) + + def __repr__(self) -> str: + return f"MarimoCells(cells={len(self._cells)}, helper={HELPER_NAME!r})" diff --git a/code_sandboxes/marimo_reactive.py b/code_sandboxes/marimo_reactive.py index 49e76dc..2be3f3d 100644 --- a/code_sandboxes/marimo_reactive.py +++ b/code_sandboxes/marimo_reactive.py @@ -84,6 +84,11 @@ def remove(self, cell_id): def code(self, cell_id): return self._code.get(cell_id) + def codes(self): + """Every registered cell's source, by id: what a second driver on + the same kernel reads back.""" + return dict(self._code) + # -- what to run ----------------------------------------------------- def plan(self, cell_id): diff --git a/code_sandboxes/marimo_sandbox.py b/code_sandboxes/marimo_sandbox.py index ced2811..89b5b22 100644 --- a/code_sandboxes/marimo_sandbox.py +++ b/code_sandboxes/marimo_sandbox.py @@ -239,6 +239,9 @@ def run_cell( started_at=time.time(), completed_at=time.time(), context_id="default", + # Named, even though it never ran: a stream consumer tags + # the refusal to the cell that was asked for. + cell_id=cell_id, ), registration=registration, ) @@ -265,7 +268,10 @@ def run_cell( result.reactions.append( Reaction(cell_id=dependent, code=dependent_code, result=dependent_result) ) - if dependent_result.code_error is not None: + # Stopped by any failure — a raise, an interrupt, a sandbox + # error, a non-zero exit — not only by a raise: nothing + # downstream runs on a state a failed cell left half-made. + if not dependent_result.success: break return run diff --git a/tests/test_marimo_cells.py b/tests/test_marimo_cells.py new file mode 100644 index 0000000..658abd5 --- /dev/null +++ b/tests/test_marimo_cells.py @@ -0,0 +1,140 @@ +# Copyright (c) 2025-2026 Datalayer, Inc. +# +# BSD 3-Clause License + +"""Marimo's cells, driven through a Jupyter-shaped `execute` on any client. + +A fake client executes in one namespace and answers Jupyter-shaped replies, +the way `CodeSandboxClient.execute` does for every variant: the driver only +ever sees replies, so this is the path the MCP server's toolset takes on a +Datalayer runtime. +""" + +from __future__ import annotations + +import contextlib +import io +import traceback + +import pytest + +marimo = pytest.importorskip("marimo") + +from code_sandboxes.marimo_cells import MarimoCells # noqa: E402 + + +class ReplyingClient: + """`execute(code)` → a Jupyter-shaped reply, from code run right here.""" + + def __init__(self, namespace=None): + self.namespace = namespace if namespace is not None else {"__name__": "__main__"} + self.executed: list[str] = [] + + def execute(self, code, timeout=None, **kwargs): + self.executed.append(code) + out, err = io.StringIO(), io.StringIO() + error = None + with contextlib.redirect_stdout(out), contextlib.redirect_stderr(err): + try: + exec(compile(code, "", "exec"), self.namespace) # noqa: S102 - the test's own code + except BaseException as raised: + error = raised + outputs = [] + if out.getvalue(): + outputs.append({"output_type": "stream", "name": "stdout", "text": out.getvalue()}) + if err.getvalue(): + outputs.append({"output_type": "stream", "name": "stderr", "text": err.getvalue()}) + if error is not None: + outputs.append( + { + "output_type": "error", + "ename": type(error).__name__, + "evalue": str(error), + "traceback": traceback.format_exception(error), + } + ) + return { + "status": "error" if error else "ok", + "execution_count": len(self.executed), + "outputs": outputs, + } + + +@pytest.fixture +def cells(): + return MarimoCells(ReplyingClient(), install_marimo=False) + + +def test_the_helper_is_installed_once_by_execute(cells): + cells.install() + cells.install() + assert sum("class _MarimoReactive" in code for code in cells._client.executed) == 1 + + +def test_running_a_cell_re_runs_what_depends_on_it(cells): + cells.run_cell("a", "x = 1") + cells.run_cell("b", "print(x + 1)") + run = cells.run_cell("a", "x = 10") + assert run.ok + assert [reaction.cell_id for reaction in run.reactions] == ["b"] + assert run.reactions[0].outputs == [{"output_type": "stream", "name": "stdout", "text": "11\n"}] + + +def test_a_failing_reaction_stops_the_chain_and_is_reported(cells): + cells.run_cell("a", "x = 1") + cells.run_cell("b", "assert x < 5") + cells.run_cell("c", "print('c reads', x)") + run = cells.run_cell("a", "x = 10") + assert run.ok + assert [(r.cell_id, r.ok) for r in run.reactions] == [("b", False)] + assert run.to_dict()["reactions"][0]["status"] == "error" + + +def test_a_cell_that_does_not_parse_never_runs(cells): + run = cells.run_cell("a", "def broken(:") + assert not run.ok + assert "SyntaxError" in run.registration["error"] + assert run.cell.outputs[0]["ename"] == "SyntaxError" + assert "a" not in cells.cells + + +def test_the_graph_and_the_plan_are_readable(cells): + cells.register_cell("a", "x = 1") + cells.register_cell("b", "y = x") + cells.register_cell("c", "z = y") + assert cells.plan("a") == ["b", "c"] + graph = cells.graph() + assert graph["cells"]["b"]["parents"] == ["a"] and graph["cells"]["b"]["children"] == ["c"] + cells.remove_cell("c") + assert cells.plan("a") == ["b"] + + +def test_run_code_is_a_cell_of_its_own(cells): + cells.run_code("base = 2") + cells.run_code("print(base * 3)") + run = cells.run_code("base = 5") + assert run.cell.cell_id == "cell-3" + assert run.reactions[0].outputs[0]["text"] == "15\n" + + +def test_a_second_driver_on_the_same_kernel_reads_the_cells_back(): + client = ReplyingClient() + first = MarimoCells(client, install_marimo=False) + first.run_cell("a", "x = 1") + first.run_cell("b", "print(x)") + second = MarimoCells(client, install_marimo=False) + assert second.cells == {"a": "x = 1", "b": "print(x)"} + run = second.run_cell("a", "x = 7") + assert run.reactions[0].outputs[0]["text"] == "7\n" + + +def test_a_missing_marimo_is_refused_when_installing_is_off(): + namespace = {"__name__": "__main__", "__builtins__": {**__builtins__, "__import__": _no_marimo}} + with pytest.raises(RuntimeError, match="install_marimo is False"): + MarimoCells(ReplyingClient(namespace), install_marimo=False).install() + + +def _no_marimo(name, *args, **kwargs): + if name == "marimo": + raise ImportError("no marimo here") + return __import__(name, *args, **kwargs) diff --git a/tests/test_marimo_sandbox.py b/tests/test_marimo_sandbox.py index d2fec8e..68c2d32 100644 --- a/tests/test_marimo_sandbox.py +++ b/tests/test_marimo_sandbox.py @@ -296,3 +296,22 @@ def test_the_client_offers_the_graph_to_a_reactive_sandbox_only(sandbox): assert plain.reactive is False with pytest.raises(TypeError, match="not reactive"): plain.plan("a") + + +def test_a_refused_cell_is_still_named_and_a_stopped_reaction_is_not_ok(sandbox): + from code_sandboxes import CodeSandboxClient, Reaction + from code_sandboxes.models import Context + + refused = sandbox.run_code("def broken(:", context=Context(id="a")) + assert refused.cell_id == "a" and refused.code_error is not None + events = list(sandbox.run_code_streaming("def broken(:", context=Context(id="a"))) + assert [e.marimo_cell_id for e in events] == ["a"] + + client = CodeSandboxClient(sandbox) + client.execute("n = 1", cell_id="p") + client.execute("n\nraise SystemExit(3)", cell_id="q") # reads n: a dependent + client.execute("print('r')", cell_id="r") # depends on nothing + reply = client.execute("n = 2", cell_id="p") + (reaction,) = reply["marimo"]["reactions"] + assert reaction["cell_id"] == "q" and reaction["status"] == "error" + assert isinstance(Reaction, type) From 01cecef5800c3025115b81d930af19653daa5170 Mon Sep 17 00:00:00 2001 From: Eric Charles Date: Sat, 26 Sep 2026 10:55:53 +0200 Subject: [PATCH 3/6] =?UTF-8?q?Docs:=20the=20Marimo=20page=20covers=20the?= =?UTF-8?q?=20implementation=20=E2=80=94=20the=20runs=20and=20their=20fiel?= =?UTF-8?q?ds,=20what=20re-runs=20and=20when=20it=20stops,=20what=20the=20?= =?UTF-8?q?graph=20refuses=20and=20what=20it=20only=20reports,=20the=20gra?= =?UTF-8?q?ph=20calls,=20the=20Jupyter-shaped=20API=20and=20its=20stream?= =?UTF-8?q?=20tags,=20the=20MarimoCells=20driver,=20and=20how=20the=20help?= =?UTF-8?q?er=20works?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/docs/providers/marimo.mdx | 173 ++++++++++++++++++++++++++++----- 1 file changed, 150 insertions(+), 23 deletions(-) diff --git a/docs/docs/providers/marimo.mdx b/docs/docs/providers/marimo.mdx index d13fc63..c32da56 100644 --- a/docs/docs/providers/marimo.mdx +++ b/docs/docs/providers/marimo.mdx @@ -18,6 +18,7 @@ requests. pass `install_marimo=False` to refuse that and fail instead. - **Parameters:** those of `jupyter-server` (`server_url`, `token`, `host`/`port`, `python_executable`), plus `install_marimo`. +- **Language:** Python only; `run_code` refuses any other `language`. ## Usage @@ -34,29 +35,90 @@ with Sandbox.create(variant="marimo") as sandbox: run.reactions[0].cell_id # "b" ``` -`run_cell` answers a `MarimoRun`: the cell's own `result`, its `registration` -(the names it defines and reads, its conflicts, a cycle) and the `reactions` -it caused, each a `CellRun`. A cell that fails stops the reaction — nothing -downstream runs on a state the failure left half-made — and `react=False` -runs the cell alone. - -The graph itself is available: `register_cell` puts a cell in it without -running it, `remove_cell` takes it out, `plan(cell_id)` lists what would -re-run, `graph()` snapshots every cell's names, neighbours, conflicts and -cycles, and `cells` is the registered source by id. +`run_cell(cell_id, code, *, react=True, timeout=None)` answers a `MarimoRun`: + +| Field | What it holds | +|---|---| +| `cell_id` | The cell that was asked for. | +| `result` | The cell's own `ExecutionResult`, with `cell_id` set and `reactions` listing what it made re-run. | +| `reactions` | The cells re-run because they depend on this one, in the order they ran, each a `CellRun` (`cell_id`, `code`, `result`). | +| `registration` | The cell's names as the graph read them — `defs`, `refs`, `conflicts`, `cycle` — or its `error` when the source does not parse. | +| `ok` | Whether the cell itself ran and raised nothing; the reactions have their own `result.success`. | + +The output handlers of `run_code` (`on_stdout`, `on_stderr`, `on_result`, +`on_error`) apply to the cell and to every reaction. + +### What re-runs, and when it stops + +- **A cell runs, then its descendants.** After the cell, every cell that + transitively reads a name it defines runs, in topological order + (`plan(cell_id)` is that list). A cell nobody reads from re-runs nothing. +- **A failure stops the reaction.** A reaction that raises, is interrupted, + or whose execution fails stops the chain: nothing downstream runs on a + state the failure left half-made, which is what Marimo does too. The + failed reaction is reported in `reactions`, not hidden; the cells after it + are simply not there. +- **A cell that fails re-runs nothing.** When the cell itself raises, its + dependents are left as they were. +- **`interrupt()` is honoured between cells.** The kernel stops the cell + that is running; a request that lands between two reactions stops the + chain before the next one. +- **`react=False` runs the cell alone**, registering it all the same, so its + dependents are stale until they run. + +### What the graph refuses, and what it only reports + +Registration is Marimo's own compiler (`marimo._ast.compiler.compile_cell`), +so a cell's `defs` and `refs` are what Marimo would compute for it. + +- **Source that does not parse is refused before running.** `registration` + holds `{"cell": id, "error": "SyntaxError: … (line n)"}`, the run's + `result.code_error` is that `SyntaxError`, and nothing was executed. The + cell is not in the graph. +- **A name defined by two cells is reported, not refused.** Marimo's editor + refuses such a notebook; here the cell registers, runs, and its + `registration["conflicts"]` names the shared definitions (`graph()` lists + them all under `conflicts`). Both definers count as parents of whatever + reads the name, so running either re-runs it. +- **A cycle is reported, not refused.** Two cells each reading what the other + defines register with `registration["cycle"] = True` (and under `cycles` + in `graph()`). Leave a cycle in the graph and the plan through it is not + well defined; remove or rewrite one side. +- **Only registered cells take part.** Code run by another means — a plain + execute request from another client, or the `envs` prelude — changes the + namespace without the graph knowing; nothing re-runs for it. + +### The graph itself + +| Call | Answers | +|---|---| +| `register_cell(cell_id, code)` | Puts the cell in the graph without running it; answers its registration (the same id again replaces the cell). | +| `remove_cell(cell_id)` | Takes the cell out; its dependents lose that parent and nothing is re-run. | +| `plan(cell_id)` | The cells that would re-run after `cell_id`, in order; `[]` for a cell that is not registered. | +| `graph()` | Every cell's `defs`, `refs`, `parents` and `children`, plus the graph's `conflicts` and `cycles`. | +| `cells` | The registered cells' source, by id. | ## Through the Jupyter-shaped API `run_code` keeps working as on every sandbox, and it is reactive too. Name the cell through the execution context — `Context(id="a")` — and the same id again -replaces that cell rather than adding a second; leave it out and each call is a -cell of its own, so an agent that never heard of cells still gets a consistent -state: what it ran earlier and that depends on what it just ran is run again. +replaces that cell rather than adding a second; leave it out (or pass the +default context) and each call is a cell of its own, named `cell-1`, `cell-2`, +…, so an agent that never heard of cells still gets a consistent state: what +it ran earlier and that depends on what it just ran is run again. The result says what happened: `cell_id` is the cell it ran as, and `reactions` lists every cell re-run because of it, in the order they ran, each a `Reaction` with the cell's `code` and its own `ExecutionResult`. A reaction -that fails stops the chain and is reported, not hidden. +that fails stops the chain and is reported, not hidden. On every other +variant `cell_id` is `None` and `reactions` is empty. + +`run_code_streaming` yields the cell's events, then each reaction's, in the +order the base streaming yields them (stdout, stderr, results, then the +error). Every event carries `marimo_cell_id` (the cell it came from) and +`marimo_reaction` (`False` for the cell that was run, `True` for a cell re-run +because of it), so a consumer that reads the stream as a Jupyter client does +can still tell the cells apart. `CodeSandboxClient` carries the same through the Jupyter protocol, so a consumer that only reads replies and streams loses nothing: @@ -74,30 +136,95 @@ reply["marimo"]["reactions"][0]["outputs"] # [{"output_type": "stream", .. ``` - `execute`, `execute_code` and `execute_code_streaming` take `cell_id`; a - reply with reactions carries them under `marimo`, each with its own - Jupyter-shaped `outputs`, `status` and `execution_count`. + reply from a reactive sandbox carries `marimo` — the `cell_id` it ran as + and the `reactions`, each with its own Jupyter-shaped `outputs`, `status` + (`"ok"` only when the reaction succeeded) and `execution_count`. - `execute_interactive` emits the reactions' outputs after the cell's own, every message tagged `metadata.marimo = {cell_id, reaction}` — the way a kernel's IOPub messages carry their parent — so a client routes each output to the cell it belongs to. -- the stream from `execute_code_streaming` yields the reactions' events after - the cell's, each tagged `marimo_cell_id` and `marimo_reaction`. - `client.reactive` says whether the sandbox keeps a graph, and `run_cell`, `register_cell`, `remove_cell`, `plan`, `graph` and `cells` pass through to it; on any other variant they raise `TypeError`. -Nothing changes on the wire: the graph is driven through ordinary execute -requests and answers on stdout, which is what keeps `jupyter-kernel-client` -the client API. - The concrete implementation is available from a top-level module: ```python from code_sandboxes.marimo_sandbox import MarimoSandbox ``` +## Reactive cells on any sandbox: `MarimoCells` + +The graph needs nothing of a kernel but execute requests and stdout, so it +does not have to be the `marimo` variant's. `MarimoCells` is the same +reactivity as a *driver* over any `CodeSandboxClient` — a Datalayer runtime, +a Kaggle kernel, a sandbox somebody else launched — through the client's +Jupyter-shaped `execute`: + +```python +from code_sandboxes import CodeSandboxClient, MarimoCells + +client = CodeSandboxClient.create(variant="datalayer", environment="python-cpu-env") +cells = MarimoCells(client) # install_marimo=True, timeout=None +cells.run_cell("a", "x = 1") +run = cells.run_cell("b", "print(x)") +cells.run_cell("a", "x = 2").reactions[0].cell_id # "b", re-run +``` + +- `install()` puts the helper in the kernel the first time it is needed: + it probes `import marimo`, installs marimo with pip when it is missing + (`install_marimo=False` raises instead), sends the helper source as one + execute request, and reads back the cells the kernel already holds. It is + idempotent, and every other call makes it first. +- `run_cell(cell_id, code, *, react=True, timeout=None)` answers a + `CellsRun`: the `cell` (a `CellReply` — `cell_id`, `code`, the reply, `ok`, + `outputs`), the `reactions` (each a `CellReply`), the `registration`, and + `ok`. The rules are the sandbox's: a refused cell never runs, a failed + reaction stops the chain. `to_dict()` is the run as plain data — `cell_id`, + `status`, `outputs`, `registration`, `reactions` — the shape a tool answers + with. +- `run_code(code)` is a cell of its own, `cell-1`, `cell-2`, …. +- `register_cell`, `remove_cell`, `plan`, `graph` and `cells` are the + sandbox's, over the client. +- **The graph lives in the kernel, not in the driver.** A second driver made + on the same sandbox — another replica answering the same session, say — + reads the registered cells back at `install()` and continues them. On a + client whose sandbox *is* a `MarimoSandbox`, both see one graph. + +This is what an MCP toolset uses to give an agent reactive cells on the +sandbox its session already holds (the Datalayer MCP Server's `marimo` +toolset, code-sandboxes#37). + +## How it works + +- **One helper, sent as source.** `code_sandboxes.marimo_reactive` holds the + source of a small class, `_MarimoReactive`, that keeps a Marimo dataflow + graph (`marimo._runtime.dataflow.DirectedGraph`) beside the kernel's + namespace. A `marimo` sandbox sends it as one execute request at `start()`; + a driver sends it at `install()`. It binds itself to `__marimo_reactive__` + and is not replaced when it is already there, which is what keeps the + graph across drivers. +- **Questions are execute requests; answers are one stdout line.** Every + graph call — `register`, `remove`, `plan`, `snapshot`, `codes` — is + `__marimo_reactive__.answer("plan", "a")` sent as code; the helper prints + `__MARIMO__` followed by the answer as base64 JSON, and the caller decodes + the stream messages it already reads. Nothing on the wire is new, which is + what keeps `jupyter-kernel-client` the client API. +- **Running is running.** A cell's execution is the ordinary `run_code` of + the `jupyter-server` sandbox — the same kernel, the same namespace, the + same output messages — and a reaction is another one. The order comes from + the graph; the results come from the kernel. +- **The same source runs in three places.** The package executes the helper + source at import time too, so its tests exercise exactly the code the + kernel runs and `local_helper()` answers graph questions without a kernel; + and `@datalayer/jupyter-react` carries the source verbatim + (`jupyter/marimo/reactive.ts`) for the browser-side components. Keep the + copies identical. + ## Management `marimo` sandboxes are kernels on a Jupyter Server, which does not tell them apart from any other kernel; `get_manager("marimo")` is the `jupyter-server` -manager answering under the `marimo` variant. +manager answering under the `marimo` variant. A sandbox's `info.variant` is +`marimo` and its `info.metadata["reactive"]` is `"marimo"`, which is how a +client tells a reactive sandbox from a plain kernel. From 494e6bde59723492b7bf6c64a84820365d407e2f Mon Sep 17 00:00:00 2001 From: Eric Charles Date: Sat, 26 Sep 2026 11:01:04 +0200 Subject: [PATCH 4/6] CI: every workflow file is .yaml, as release.yaml already was; callers and the contributing page follow --- .github/workflows/{build.yml => build.yaml} | 2 +- .../{environments-live.yml => environments-live.yaml} | 0 .../{py-code-style.yml => py-code-style.yaml} | 10 +++++----- .github/workflows/{py-tests.yml => py-tests.yaml} | 10 +++++----- .github/workflows/{py-typing.yml => py-typing.yaml} | 10 +++++----- .../{reusable-python.yml => reusable-python.yaml} | 0 CHANGELOG.md | 6 +++++- docs/docs/contribute/index.mdx | 10 +++++----- 8 files changed, 26 insertions(+), 22 deletions(-) rename .github/workflows/{build.yml => build.yaml} (92%) rename .github/workflows/{environments-live.yml => environments-live.yaml} (100%) rename .github/workflows/{py-code-style.yml => py-code-style.yaml} (76%) rename .github/workflows/{py-tests.yml => py-tests.yaml} (77%) rename .github/workflows/{py-typing.yml => py-typing.yaml} (76%) rename .github/workflows/{reusable-python.yml => reusable-python.yaml} (100%) diff --git a/.github/workflows/build.yml b/.github/workflows/build.yaml similarity index 92% rename from .github/workflows/build.yml rename to .github/workflows/build.yaml index d21ae29..599603e 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yaml @@ -13,7 +13,7 @@ jobs: strategy: matrix: python-version: ['3.11', '3.12'] - uses: ./.github/workflows/reusable-python.yml + uses: ./.github/workflows/reusable-python.yaml with: python-version: ${{ matrix.python-version }} install-extras: test,typing,datalayer diff --git a/.github/workflows/environments-live.yml b/.github/workflows/environments-live.yaml similarity index 100% rename from .github/workflows/environments-live.yml rename to .github/workflows/environments-live.yaml diff --git a/.github/workflows/py-code-style.yml b/.github/workflows/py-code-style.yaml similarity index 76% rename from .github/workflows/py-code-style.yml rename to .github/workflows/py-code-style.yaml index cabbd6c..7575a3a 100644 --- a/.github/workflows/py-code-style.yml +++ b/.github/workflows/py-code-style.yaml @@ -13,8 +13,8 @@ on: - '.pre-commit-config.yaml' - 'ruff.toml' - '.ruff.toml' - - '.github/workflows/py-code-style.yml' - - '.github/workflows/reusable-python.yml' + - '.github/workflows/py-code-style.yaml' + - '.github/workflows/reusable-python.yaml' pull_request: branches: - main @@ -27,8 +27,8 @@ on: - '.pre-commit-config.yaml' - 'ruff.toml' - '.ruff.toml' - - '.github/workflows/py-code-style.yml' - - '.github/workflows/reusable-python.yml' + - '.github/workflows/py-code-style.yaml' + - '.github/workflows/reusable-python.yaml' workflow_dispatch: concurrency: @@ -37,7 +37,7 @@ concurrency: jobs: check-code-style: - uses: ./.github/workflows/reusable-python.yml + uses: ./.github/workflows/reusable-python.yaml with: python-version: '3.11' extra-packages: pre-commit diff --git a/.github/workflows/py-tests.yml b/.github/workflows/py-tests.yaml similarity index 77% rename from .github/workflows/py-tests.yml rename to .github/workflows/py-tests.yaml index ad325c4..b212c1f 100644 --- a/.github/workflows/py-tests.yml +++ b/.github/workflows/py-tests.yaml @@ -11,8 +11,8 @@ on: - 'setup.py' - 'requirements*.txt' - '**/*.py' - - '.github/workflows/py-tests.yml' - - '.github/workflows/reusable-python.yml' + - '.github/workflows/py-tests.yaml' + - '.github/workflows/reusable-python.yaml' pull_request: branches: - main @@ -23,8 +23,8 @@ on: - 'setup.py' - 'requirements*.txt' - '**/*.py' - - '.github/workflows/py-tests.yml' - - '.github/workflows/reusable-python.yml' + - '.github/workflows/py-tests.yaml' + - '.github/workflows/reusable-python.yaml' workflow_dispatch: @@ -34,7 +34,7 @@ jobs: max-parallel: 1 matrix: python-version: ['3.10', '3.11', '3.12', '3.13'] - uses: ./.github/workflows/reusable-python.yml + uses: ./.github/workflows/reusable-python.yaml with: python-version: ${{ matrix.python-version }} install-extras: test,datalayer diff --git a/.github/workflows/py-typing.yml b/.github/workflows/py-typing.yaml similarity index 76% rename from .github/workflows/py-typing.yml rename to .github/workflows/py-typing.yaml index f5d68a2..669faf7 100644 --- a/.github/workflows/py-typing.yml +++ b/.github/workflows/py-typing.yaml @@ -12,8 +12,8 @@ on: - 'setup.py' - 'mypy.ini' - '.mypy.ini' - - '.github/workflows/py-typing.yml' - - '.github/workflows/reusable-python.yml' + - '.github/workflows/py-typing.yaml' + - '.github/workflows/reusable-python.yaml' pull_request: branches: - main @@ -25,8 +25,8 @@ on: - 'setup.py' - 'mypy.ini' - '.mypy.ini' - - '.github/workflows/py-typing.yml' - - '.github/workflows/reusable-python.yml' + - '.github/workflows/py-typing.yaml' + - '.github/workflows/reusable-python.yaml' workflow_dispatch: concurrency: @@ -35,7 +35,7 @@ concurrency: jobs: type-check: - uses: ./.github/workflows/reusable-python.yml + uses: ./.github/workflows/reusable-python.yaml with: python-version: '3.12' install-extras: typing diff --git a/.github/workflows/reusable-python.yml b/.github/workflows/reusable-python.yaml similarity index 100% rename from .github/workflows/reusable-python.yml rename to .github/workflows/reusable-python.yaml diff --git a/CHANGELOG.md b/CHANGELOG.md index 2568e99..6acd2f0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,6 +10,10 @@ ## 1.9.39 +- The GitHub workflow files are all `.yaml` now (`build`, `py-tests`, + `py-code-style`, `py-typing`, `reusable-python`, `environments-live`), the + spelling `release.yaml` already had; the reusable workflow's callers and + the contributing page follow. - **Marimo reactivity through the Jupyter-shaped API** (#37). `run_code` names its cell through the execution context (`Context(id=...)`; the same id replaces the cell) and the result says what happened: `cell_id`, and @@ -353,7 +357,7 @@ that `stop()` skipped entirely, orphaned for good. Fixed identically in `daytona_sandbox.py`, `modal_sandbox.py` and `e2b_sandbox.py`. - **A nightly live matrix for the three managed builders** - (`.github/workflows/environments-live.yml`, PLAN_ENV.md E2-11). Builds a + (`.github/workflows/environments-live.yaml`, PLAN_ENV.md E2-11). Builds a real artifact on each real provider from a real, hash-verified lock, launches a sandbox from it, and runs the formal core tier — opening (or commenting on) an issue naming the provider and the check on a genuine diff --git a/docs/docs/contribute/index.mdx b/docs/docs/contribute/index.mdx index 2928614..5de6d33 100644 --- a/docs/docs/contribute/index.mdx +++ b/docs/docs/contribute/index.mdx @@ -42,14 +42,14 @@ Optional test variables: ## CI Workflows This repository uses a reusable GitHub Actions workflow at -`.github/workflows/reusable-python.yml`. +`.github/workflows/reusable-python.yaml`. Workflows that call it: -- `.github/workflows/build.yml` -- `.github/workflows/py-tests.yml` -- `.github/workflows/py-code-style.yml` -- `.github/workflows/py-typing.yml` +- `.github/workflows/build.yaml` +- `.github/workflows/py-tests.yaml` +- `.github/workflows/py-code-style.yaml` +- `.github/workflows/py-typing.yaml` Reusable workflow inputs: From d31290ab85b6d2dd89bb8a286f2124572c579742 Mon Sep 17 00:00:00 2001 From: Eric Charles Date: Sat, 26 Sep 2026 11:28:31 +0200 Subject: [PATCH 5/6] docs --- docs/docs/index.mdx | 4 +--- docs/docusaurus.config.js | 6 ++++++ 2 files changed, 7 insertions(+), 3 deletions(-) diff --git a/docs/docs/index.mdx b/docs/docs/index.mdx index 192dbea..7941022 100644 --- a/docs/docs/index.mdx +++ b/docs/docs/index.mdx @@ -1,14 +1,12 @@ --- sidebar_position: 0 -title: Code Sandboxes +title: 📦 Code Sandboxes slug: / --- import DocCardList from '@theme/DocCardList'; import { useDocsSidebar } from '@docusaurus/plugin-content-docs/client'; -📦 Code Sandboxes - **Code Sandboxes** is a Python package for creating safe, isolated environments where AI systems can write, run, and test code without affecting the real world or the user's device. ## Package Scope diff --git a/docs/docusaurus.config.js b/docs/docusaurus.config.js index c18c593..cc2290b 100644 --- a/docs/docusaurus.config.js +++ b/docs/docusaurus.config.js @@ -42,6 +42,12 @@ module.exports = { position: 'left', label: 'Guide', }, + { + type: 'doc', + docId: 'provider-ingress/index', + position: 'left', + label: 'Provider Ingress', + }, { type: 'doc', docId: 'install/index', From 413af0cfcc25871ee3adb8d5d9e700381bdad68a Mon Sep 17 00:00:00 2001 From: Eric Charles Date: Sat, 26 Sep 2026 11:32:45 +0200 Subject: [PATCH 6/6] docs --- docs/docs/index.mdx | 4 +++- .../docs/{provider-ingress.mdx => provider-ingress/index.mdx} | 0 2 files changed, 3 insertions(+), 1 deletion(-) rename docs/docs/{provider-ingress.mdx => provider-ingress/index.mdx} (100%) diff --git a/docs/docs/index.mdx b/docs/docs/index.mdx index 7941022..93dfa01 100644 --- a/docs/docs/index.mdx +++ b/docs/docs/index.mdx @@ -1,12 +1,14 @@ --- sidebar_position: 0 -title: 📦 Code Sandboxes +title: Code Sandboxes slug: / --- import DocCardList from '@theme/DocCardList'; import { useDocsSidebar } from '@docusaurus/plugin-content-docs/client'; +# 📦 Code Sandboxes + **Code Sandboxes** is a Python package for creating safe, isolated environments where AI systems can write, run, and test code without affecting the real world or the user's device. ## Package Scope diff --git a/docs/docs/provider-ingress.mdx b/docs/docs/provider-ingress/index.mdx similarity index 100% rename from docs/docs/provider-ingress.mdx rename to docs/docs/provider-ingress/index.mdx