From 7e770483500494c1a4fec3e9d097e475c736751f Mon Sep 17 00:00:00 2001 From: Lalit Gupta Date: Sat, 25 Jul 2026 02:12:34 +0530 Subject: [PATCH 1/4] feat: add sandbox_id to Audio.generate_transcript for Whisper sandbox routing --- videodb/audio.py | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/videodb/audio.py b/videodb/audio.py index 9dd6987..3aa4d31 100644 --- a/videodb/audio.py +++ b/videodb/audio.py @@ -138,6 +138,7 @@ def generate_transcript( self, force: bool = None, language_code: str = None, + sandbox_id: str = None, ) -> dict: """Generate transcript for the audio. @@ -146,15 +147,21 @@ def generate_transcript( Use ISO 639-1 codes (e.g., "en", "hi", "fr") or regional variants with underscores (e.g., "en_us", "en_uk", "en_au"). Defaults to "en_us" if not specified. + :param str sandbox_id: Optional sandbox ID to run transcription on a + Whisper-backed sandbox. When omitted, the default hosted + transcription path is used. :return: Success dict if transcript generated or already exists :rtype: dict """ + data = { + "force": True if force else False, + "language_code": language_code, + } + if sandbox_id: + data["sandbox_id"] = sandbox_id transcript_data = self._connection.post( path=f"{ApiPath.audio}/{self.id}/{ApiPath.transcription}", - data={ - "force": True if force else False, - "language_code": language_code, - }, + data=data, ) transcript = transcript_data.get("word_timestamps", []) if transcript: From 9442bffd5c47a87582979df5d20b592471210f8b Mon Sep 17 00:00:00 2001 From: Lalit Gupta Date: Sat, 25 Jul 2026 13:31:36 +0530 Subject: [PATCH 2/4] fix(sandbox): validate models/model_categories on create; clarify stop grace is reserved --- videodb/client.py | 12 ++++++++++-- videodb/sandbox.py | 4 +++- 2 files changed, 13 insertions(+), 3 deletions(-) diff --git a/videodb/client.py b/videodb/client.py index e53374d..7ca4368 100644 --- a/videodb/client.py +++ b/videodb/client.py @@ -339,11 +339,19 @@ def create_sandbox( :param str tier: Sandbox tier — "small" or "medium" (default: server decides) :param str name: Human-readable name (auto-generated if not provided) :param str callback_url: URL to receive sandbox lifecycle webhooks - :param list[str] model_categories: Model categories to prepare for this sandbox, e.g. ``["vlm", "image_generation"]`` (optional) - :param list[str] models: Specific model names to prepare for this sandbox (optional) + :param list[str] model_categories: Model categories to prepare for this sandbox, e.g. ``["vlm", "image_generation"]`` + :param list[str] models: Specific model names to prepare for this sandbox :return: :class:`Sandbox ` object in provisioning state :rtype: :class:`videodb.sandbox.Sandbox` + :raises ValueError: If neither ``models`` nor ``model_categories`` is provided + + .. note:: At least one of ``models`` or ``model_categories`` is required. """ + if not models and not model_categories: + raise ValueError( + "At least one of 'models' or 'model_categories' is required " + "to create a sandbox." + ) data = self.post( path=ApiPath.sandbox, data={ diff --git a/videodb/sandbox.py b/videodb/sandbox.py index 8b3fabf..6e83360 100644 --- a/videodb/sandbox.py +++ b/videodb/sandbox.py @@ -91,7 +91,9 @@ def wait_for_ready(self, timeout=300, interval=5): def stop(self, grace=True): """Stop this sandbox. - :param bool grace: Wait for running jobs to finish before teardown (default True) + :param bool grace: Reserved. Sandbox teardown is currently always + graceful (running jobs finish before compute is released); this + flag has no effect yet and is kept for forward compatibility. :return: self """ data = self._connection.post( From 05b586504d9b1819def52cc00faec48d284510f3 Mon Sep 17 00:00:00 2001 From: Lalit Gupta Date: Wed, 26 Aug 2026 12:53:49 +0530 Subject: [PATCH 3/4] feat(rtstream): support cua analyzer + trigger on understand() Adds a top-level trigger param (interval|on_demand) to RTStream.understand() and documents the cua (computer-use) analyzer with past_window. The cua analyzer and past_window already flow through the analyzers list. --- videodb/rtstream.py | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/videodb/rtstream.py b/videodb/rtstream.py index 93f1928..89ceb5d 100644 --- a/videodb/rtstream.py +++ b/videodb/rtstream.py @@ -1100,18 +1100,24 @@ def understand( analyzers: List[Dict] = None, store: bool = True, ws_connection_id: str = None, + trigger: str = None, ) -> "RTStreamUnderstanding": - """Start a continuous VLM understanding job on the stream. + """Start a continuous understanding job on the stream. Understanding is independent of indexing: it produces VLM output per stream window and (when ``store=True``) persists it so it can be indexed - later. Initial support is one ``vlm`` analyzer with time segmentation. + later. Supports one ``vlm`` or ``cua`` (computer-use) analyzer with time + segmentation. :param dict segmentation: Time segmentation, e.g. ``{"type": "time", "window": "10s"}`` - :param list analyzers: Exactly one VLM analyzer spec, e.g. + :param list analyzers: Exactly one analyzer spec, e.g. ``[{"type": "vlm", "name": "scene", "sampling": {"frame_count": 5}, "config": {"prompt": "...", "model": "basic"}}]`` + For a computer-use agent, ``[{"type": "cua", "name": "action", "sampling": {"frame_count": 1}, "config": {"prompt": "", "past_window": {"frames": 3, "actions": 5}}}]`` + (a ``cua`` analyzer defaults its model to Holo) :param bool store: Persist output for later indexing (default: True) :param str ws_connection_id: WebSocket connection ID for real-time updates (optional) + :param str trigger: ``"interval"`` (default) samples on a fixed cadence; + ``"on_demand"`` produces one output per external trigger (CUA loop) :return: The understanding job, :class:`RTStreamUnderstanding ` object :rtype: :class:`videodb.rtstream.RTStreamUnderstanding` """ @@ -1120,6 +1126,8 @@ def understand( "analyzers": analyzers or [], "store": store, } + if trigger: + data["trigger"] = trigger if ws_connection_id: data["ws_connection_id"] = ws_connection_id understanding_data = self._connection.post( From 3a454669e44886d30b344e7ce5cdc434e4d9e6a8 Mon Sep 17 00:00:00 2001 From: Lalit Gupta Date: Thu, 27 Aug 2026 14:32:11 +0530 Subject: [PATCH 4/4] =?UTF-8?q?feat(rtstream):=20RTStreamUnderstanding.nex?= =?UTF-8?q?t(step=5Fid)=20=E2=80=94=20advance=20on-demand=20CUA=20loop?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit understanding.next(step_id) → POST /rtstream/{id}/understand/{und}/next. The per-action poke of the on-demand (trigger=on_demand) computer-use loop: after the client executes an action and the screen settles, next() has the model analyze the freshest frame and emit the next action. understanding_id is supplied automatically (self.id); the worker is idempotent on the monotonic step_id, so a duplicate/late next is safely ignored. No-op on an interval understanding. + ApiPath.next. Part of the on-demand CUA chain (ENG-1719); Server route rtstream-cua-next / #980. --- videodb/_constants.py | 1 + videodb/rtstream.py | 22 ++++++++++++++++++++++ 2 files changed, 23 insertions(+) diff --git a/videodb/_constants.py b/videodb/_constants.py index 1a1cde8..251e896 100644 --- a/videodb/_constants.py +++ b/videodb/_constants.py @@ -100,6 +100,7 @@ class ApiPath: indexes = "indexes" records = "records" understand = "understand" + next = "next" search = "search" ask = "ask" semantic_search = "semantic-search" diff --git a/videodb/rtstream.py b/videodb/rtstream.py index 89ceb5d..15ba168 100644 --- a/videodb/rtstream.py +++ b/videodb/rtstream.py @@ -453,6 +453,28 @@ def stop(self): ) self.status = "stopped" + def next(self, step_id: int): + """Advance an on-demand (CUA) understanding by one step. + + The per-action poke of the computer-use loop: after the client executes an + action and the screen settles, call ``next`` to have the model analyze the + freshest frame and emit the next action. Only meaningful for an understanding + created with ``trigger="on_demand"``; on an interval understanding the model + runs on its own clock and this is a no-op. + + ``step_id`` is a monotonic counter the client increments per action — the worker + is idempotent on it, so a duplicate or late ``next`` is safely ignored. The + understanding id is supplied automatically. + + :param int step_id: Monotonic step counter (increment once per executed action) + :return: Accepted acknowledgement ``{understanding_id, step_id}`` + :rtype: dict + """ + return self._connection.post( + f"{ApiPath.rtstream}/{self.rtstream_id}/{ApiPath.understand}/{self.id}/{ApiPath.next}", + data={"step_id": step_id}, + ) + def get_records( self, start: float,