diff --git a/nemo_curator/stages/video/caption/caption_generation.py b/nemo_curator/stages/video/caption/caption_generation.py
index 7916ae5..4ee8537 100644
--- a/nemo_curator/stages/video/caption/caption_generation.py
+++ b/nemo_curator/stages/video/caption/caption_generation.py
@@ -24,6 +24,7 @@ from nemo_curator.models.nemotron_h_vl import NemotronHVL
from nemo_curator.models.qwen_vl import _QWEN_VARIANTS_INFO, QwenVL
from nemo_curator.stages.base import ProcessingStage
from nemo_curator.stages.resources import Resources
+from nemo_curator.stages.video.caption.video_scripting import VIDEO_SCRIPT_METADATA_KEY, build_video_script
from nemo_curator.tasks.video import Video, VideoTask
@@ -46,6 +47,8 @@ class CaptionGenerationStage(ProcessingStage[VideoTask, VideoTask]):
verbose: bool = False
generate_stage2_caption: bool = False
stage2_prompt_text: str | None = None
+ build_entity_script: bool = False
+ entity_script_max_entities: int = 12
name: str = "caption_generation"
def inputs(self) -> tuple[list[str], list[str]]:
@@ -136,6 +139,16 @@ class CaptionGenerationStage(ProcessingStage[VideoTask, VideoTask]):
self._assign_captions(video, mapping, enumerate(captions))
+ if self.build_entity_script:
+ # Aggregate the freshly generated per-window captions into a single
+ # entity-anchored script so downstream stages can reason across
+ # segments with a shared entity prior.
+ task._metadata[VIDEO_SCRIPT_METADATA_KEY] = build_video_script(
+ video,
+ model_variant=self.model_variant,
+ max_entities=self.entity_script_max_entities,
+ )
+
return task
def _assign_captions(
diff --git a/nemo_curator/stages/video/caption/video_scripting.py b/nemo_curator/stages/video/caption/video_scripting.py
new file mode 100644
index 0000000..cfca134
--- /dev/null
+++ b/nemo_curator/stages/video/caption/video_scripting.py
@@ -0,0 +1,313 @@
+# Copyright (c) 2025, NVIDIA CORPORATION. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+"""Entity-anchored video scripting.
+
+Aggregates the per-window captions already produced by the captioning stages of
+a :class:`~nemo_curator.tasks.video.Video` into a single *structured script*: a
+short summary, a global list of recurring entities, and segment-wise
+descriptions anchored to those entities.
+
+The decoupled, per-clip captioning paradigm severs associations between an
+entity and its appearances across segments β the same person or object is
+described from scratch in every clip, with no shared referent. Building a global
+entity list *first* and then tagging which segments each entity appears in
+restores cross-segment referential consistency, so downstream curation (QA
+synthesis, dedup, retrieval) can reason over the video as a whole rather than a
+bag of independent clip captions.
+
+Adapted from the "Entity-Anchored Video Scripting" mechanism in
+*OmniVideo-100K: A Dataset for Audio-Visual Reasoning through Structured Scripts
+and Evidence Chains* (https://arxiv.org/abs/2606.14702v1). This module delivers
+the structured-script representation deterministically from existing captions;
+the paper's model-driven clue-guided QA generation is intentionally out of
+scope.
+"""
+
+import re
+from dataclasses import dataclass, field
+from typing import Any
+
+from loguru import logger
+
+from nemo_curator.backends.base import WorkerMetadata
+from nemo_curator.stages.base import ProcessingStage
+from nemo_curator.tasks.video import Video, VideoTask
+
+# Metadata key under which the built script is stored on the task.
+VIDEO_SCRIPT_METADATA_KEY = "video_script"
+
+# Common function words that should never be treated as entities.
+_STOPWORD_TEXT = (
+ "a an the this that these those is are was were be been being am "
+ "and or but nor so yet for of to in on at by with from into onto over under "
+ "as it its their there here then than up down out off "
+ "he she they we you i him her them us his hers our your my mine ours theirs "
+ "who whom whose which what when where why how "
+ "very more most some any all both each few many much several "
+ "not no only also just even still about around through during while "
+ "video clip scene shows showing shown depicts depicting features featuring "
+ "seen visible appears appear appearing background foreground frame camera"
+)
+_STOPWORDS: frozenset[str] = frozenset(_STOPWORD_TEXT.split())
+
+# A candidate entity token: a word of 3+ characters, optionally hyphenated.
+_WORD_RE = re.compile(r"[A-Za-z][A-Za-z'-]{2,}")
+
+
+@dataclass
+class EntityMention:
+ """A single entity surfaced across one or more segments of a video.
+
+ Attributes:
+ name: Normalized (lowercase) surface form of the entity.
+ segment_indices: Sorted segment indices in which the entity appears.
+ count: Total number of times the entity is mentioned across all segments.
+ recurring: True if the entity appears in two or more distinct segments;
+ these are the cross-segment "anchors" that the global entity list
+ is built to preserve.
+ """
+
+ name: str
+ segment_indices: list[int] = field(default_factory=list)
+ count: int = 0
+
+ @property
+ def recurring(self) -> bool:
+ return len(self.segment_indices) >= 2 # noqa: PLR2004
+
+ def to_dict(self) -> dict[str, Any]:
+ return {
+ "name": self.name,
+ "segment_indices": list(self.segment_indices),
+ "count": self.count,
+ "recurring": self.recurring,
+ }
+
+
+@dataclass
+class VideoSegment:
+ """One segment of the script, corresponding to a single clip.
+
+ Attributes:
+ index: Position of the segment in the video (0-based).
+ span: ``(start, end)`` time span of the underlying clip in seconds.
+ description: Concatenated window caption text for the clip.
+ entities: Names of the global entities that appear in this segment.
+ """
+
+ index: int
+ span: tuple[float, float]
+ description: str
+ entities: list[str] = field(default_factory=list)
+
+ def to_dict(self) -> dict[str, Any]:
+ return {
+ "index": self.index,
+ "span": list(self.span),
+ "description": self.description,
+ "entities": list(self.entities),
+ }
+
+
+@dataclass
+class VideoScript:
+ """Structured, entity-anchored script for a whole video.
+
+ Attributes:
+ summary: Short natural-language summary leading with recurring entities.
+ entities: Global entity list, ordered by cross-segment salience.
+ segments: Per-clip segment descriptions, each tagged with its entities.
+ """
+
+ summary: str
+ entities: list[EntityMention] = field(default_factory=list)
+ segments: list[VideoSegment] = field(default_factory=list)
+
+ @property
+ def recurring_entities(self) -> list[EntityMention]:
+ return [e for e in self.entities if e.recurring]
+
+ def to_dict(self) -> dict[str, Any]:
+ return {
+ "summary": self.summary,
+ "entities": [e.to_dict() for e in self.entities],
+ "segments": [s.to_dict() for s in self.segments],
+ }
+
+
+def _segment_text(video: Video, clip_index: int, model_variant: str, *, use_enhanced: bool) -> str:
+ """Join the caption text of every window in a clip into one description."""
+ clip = video.clips[clip_index]
+ parts: list[str] = []
+ for window in clip.windows:
+ caption = None
+ if use_enhanced and window.enhanced_caption:
+ # Enhanced captions are keyed by the LM, not the captioning model;
+ # take any available enhancement, falling back to the raw caption.
+ caption = next(iter(window.enhanced_caption.values()), None)
+ if caption is None:
+ caption = window.caption.get(model_variant)
+ if caption:
+ parts.append(caption.strip())
+ return " ".join(p for p in parts if p)
+
+
+def _extract_entities(text: str) -> list[str]:
+ """Return the candidate entity tokens in *text*, lowercased, in order."""
+ tokens = []
+ for match in _WORD_RE.finditer(text):
+ token = match.group(0).lower().strip("-'")
+ if len(token) >= 3 and token not in _STOPWORDS: # noqa: PLR2004
+ tokens.append(token)
+ return tokens
+
+
+def build_video_script(
+ video: Video,
+ model_variant: str = "qwen2.5",
+ *,
+ use_enhanced: bool = False,
+ max_entities: int = 12,
+ max_summary_segments: int = 3,
+) -> VideoScript:
+ """Build an entity-anchored :class:`VideoScript` from a captioned video.
+
+ The video's clips are treated as ordered segments. Candidate entities are
+ mined from every segment's caption text and merged into a single global
+ list; each entity records which segments it appears in. Entities that recur
+ across multiple segments are promoted to the front of the list, acting as
+ the global prior that re-links references the per-clip captions had severed.
+
+ Args:
+ video: A video whose windows already carry captions.
+ model_variant: Caption key to read window captions from.
+ use_enhanced: Prefer ``enhanced_caption`` text when available.
+ max_entities: Cap on the size of the returned global entity list.
+ max_summary_segments: Number of leading segments to fold into the summary.
+
+ Returns:
+ A :class:`VideoScript` with a summary, global entity list, and segments.
+ """
+ segments: list[VideoSegment] = []
+ # name -> (segment indices set, total count)
+ entity_segments: dict[str, set[int]] = {}
+ entity_counts: dict[str, int] = {}
+
+ for clip_index in range(len(video.clips)):
+ description = _segment_text(video, clip_index, model_variant, use_enhanced=use_enhanced)
+ tokens = _extract_entities(description)
+ for token in tokens:
+ entity_segments.setdefault(token, set()).add(clip_index)
+ entity_counts[token] = entity_counts.get(token, 0) + 1
+ segments.append(
+ VideoSegment(
+ index=clip_index,
+ span=video.clips[clip_index].span,
+ description=description,
+ entities=[],
+ )
+ )
+
+ # Rank entities by cross-segment salience first (recurrence is the signal the
+ # paper anchors on), then total frequency, then name for determinism.
+ ranked = sorted(
+ entity_segments.keys(),
+ key=lambda name: (-len(entity_segments[name]), -entity_counts[name], name),
+ )
+ selected = ranked[:max_entities]
+
+ entities = [
+ EntityMention(
+ name=name,
+ segment_indices=sorted(entity_segments[name]),
+ count=entity_counts[name],
+ )
+ for name in selected
+ ]
+
+ # Anchor each segment to the global entities it contains, preserving the
+ # global ordering so cross-segment anchors come first in every segment.
+ selected_set = set(selected)
+ for entity in entities:
+ for seg_index in entity.segment_indices:
+ segments[seg_index].entities.append(entity.name)
+ for segment in segments:
+ segment.entities = [name for name in selected if name in set(segment.entities) & selected_set]
+
+ summary = _build_summary(entities, segments, max_summary_segments=max_summary_segments)
+ return VideoScript(summary=summary, entities=entities, segments=segments)
+
+
+def _build_summary(
+ entities: list[EntityMention],
+ segments: list[VideoSegment],
+ *,
+ max_summary_segments: int,
+) -> str:
+ """Compose a short summary that leads with the recurring entity prior."""
+ recurring = [e.name for e in entities if e.recurring]
+ lead = ""
+ if recurring:
+ lead = "Recurring entities: " + ", ".join(recurring[:5]) + ". "
+
+ described = [s.description for s in segments[:max_summary_segments] if s.description]
+ body = " ".join(described)
+ return (lead + body).strip()
+
+
+@dataclass
+class VideoScriptingStage(ProcessingStage[VideoTask, VideoTask]):
+ """Stage that builds an entity-anchored script from captioned video clips.
+
+ Slots in after a caption-generation/enhancement stage. It reads the captions
+ already attached to each window, aggregates them into a structured script
+ with a global entity list, and stores the result on the task metadata under
+ :data:`VIDEO_SCRIPT_METADATA_KEY`. No model is loaded β the stage operates on
+ existing caption text.
+ """
+
+ model_variant: str = "qwen2.5"
+ use_enhanced: bool = False
+ max_entities: int = 12
+ max_summary_segments: int = 3
+ verbose: bool = False
+ name: str = "video_scripting"
+
+ def inputs(self) -> tuple[list[str], list[str]]:
+ return ["data"], ["clips"]
+
+ def outputs(self) -> tuple[list[str], list[str]]:
+ return ["data"], ["clips"]
+
+ def setup(self, worker_metadata: WorkerMetadata | None = None) -> None: # noqa: ARG002
+ # No model or remote resources are required.
+ return
+
+ def process(self, task: VideoTask) -> VideoTask:
+ script = build_video_script(
+ task.data,
+ model_variant=self.model_variant,
+ use_enhanced=self.use_enhanced,
+ max_entities=self.max_entities,
+ max_summary_segments=self.max_summary_segments,
+ )
+ task._metadata[VIDEO_SCRIPT_METADATA_KEY] = script
+ if self.verbose:
+ logger.info(
+ f"Built video script for {task.data.input_path}: "
+ f"{len(script.segments)} segments, {len(script.entities)} entities "
+ f"({len(script.recurring_entities)} recurring)"
+ )
+ return task
diff --git a/tests/stages/video/caption/test_video_scripting.py b/tests/stages/video/caption/test_video_scripting.py
new file mode 100644
index 0000000..3f405cb
--- /dev/null
+++ b/tests/stages/video/caption/test_video_scripting.py
@@ -0,0 +1,163 @@
+# Copyright (c) 2025, NVIDIA CORPORATION. All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+# http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+from __future__ import annotations
+
+import pathlib
+from unittest.mock import Mock
+from uuid import uuid4
+
+# Import from the existing (non-new) call-site module to exercise the wiring.
+from nemo_curator.stages.video.caption.caption_generation import CaptionGenerationStage
+from nemo_curator.stages.video.caption.video_scripting import (
+ VIDEO_SCRIPT_METADATA_KEY,
+ VideoScript,
+ VideoScriptingStage,
+ build_video_script,
+)
+from nemo_curator.tasks.video import Clip, Video, VideoTask, _Window
+
+
+def _video_with_captions(captions: list[list[str]], model_variant: str = "qwen2.5") -> Video:
+ """Build a Video whose clips/windows already carry the given captions.
+
+ Args:
+ captions: One list of window captions per clip.
+ """
+ video = Video(input_video=pathlib.Path("test.mp4"))
+ clips = []
+ for clip_idx, window_caps in enumerate(captions):
+ clip = Clip(uuid=uuid4(), source_video=f"c{clip_idx}.mp4", span=(float(clip_idx), float(clip_idx) + 1.0))
+ clip.windows = [_Window(start_frame=0, end_frame=10, caption={model_variant: cap}) for cap in window_caps]
+ clips.append(clip)
+ video.clips = clips
+ return video
+
+
+class TestBuildVideoScript:
+ def test_builds_segments_per_clip(self):
+ video = _video_with_captions([["A red car drives down a street."], ["The car stops at a light."]])
+ script = build_video_script(video)
+
+ assert isinstance(script, VideoScript)
+ assert len(script.segments) == 2
+ assert script.segments[0].span == (0.0, 1.0)
+ assert "red car drives" in script.segments[0].description
+
+ def test_global_entity_list_links_segments(self):
+ # "car" appears in both clips -> it must be a recurring cross-segment anchor.
+ video = _video_with_captions([["A red car drives down a street."], ["The car stops at a light."]])
+ script = build_video_script(video)
+
+ car = next(e for e in script.entities if e.name == "car")
+ assert car.segment_indices == [0, 1]
+ assert car.recurring is True
+ assert car.count == 2
+ assert "car" in {e.name for e in script.recurring_entities}
+
+ def test_segments_anchored_to_global_entities(self):
+ video = _video_with_captions([["A red car drives down a street."], ["The car stops at a light."]])
+ script = build_video_script(video)
+
+ # Both segments reference the shared "car" entity.
+ assert "car" in script.segments[0].entities
+ assert "car" in script.segments[1].entities
+
+ def test_summary_leads_with_recurring_entities(self):
+ video = _video_with_captions([["A dog runs in a park."], ["The dog chases a ball."]])
+ script = build_video_script(video)
+ assert script.summary.startswith("Recurring entities: dog")
+
+ def test_stopwords_excluded_from_entities(self):
+ video = _video_with_captions([["The the and of in on."]])
+ script = build_video_script(video)
+ assert script.entities == []
+
+ def test_max_entities_caps_list(self):
+ video = _video_with_captions([["alpha beta gamma delta epsilon zeta"]])
+ script = build_video_script(video, max_entities=3)
+ assert len(script.entities) == 3
+
+ def test_use_enhanced_prefers_enhanced_caption(self):
+ video = Video(input_video=pathlib.Path("t.mp4"))
+ clip = Clip(uuid=uuid4(), source_video="c.mp4", span=(0.0, 1.0))
+ window = _Window(start_frame=0, end_frame=10, caption={"qwen2.5": "raw helicopter"})
+ window.enhanced_caption = {"qwen_lm": "enhanced submarine"}
+ clip.windows = [window]
+ video.clips = [clip]
+
+ script = build_video_script(video, use_enhanced=True)
+ assert "submarine" in script.segments[0].description
+ assert "helicopter" not in script.segments[0].description
+
+
+class TestVideoScriptingStage:
+ def test_process_attaches_script_to_metadata(self):
+ stage = VideoScriptingStage()
+ stage.setup()
+ video = _video_with_captions([["A red car on the road."], ["The car turns left."]])
+ task = VideoTask(dataset_name="test", data=video)
+
+ result = stage.process(task)
+
+ assert result is task
+ script = result._metadata[VIDEO_SCRIPT_METADATA_KEY]
+ assert isinstance(script, VideoScript)
+ assert len(script.segments) == 2
+
+ def test_stage_io_contract_matches_caption_stage(self):
+ # The scripting stage must share the VideoTask I/O contract so it can
+ # drop in after CaptionGenerationStage in the same pipeline.
+ assert VideoScriptingStage().inputs() == CaptionGenerationStage().inputs()
+ assert VideoScriptingStage().outputs() == CaptionGenerationStage().outputs()
+
+
+class TestCaptionGenerationScriptHook:
+ """Exercise the wiring edit made in CaptionGenerationStage.process()."""
+
+ def _make_task(self) -> VideoTask:
+ video = Video(input_video=pathlib.Path("test.mp4"))
+ clip1 = Clip(uuid=uuid4(), source_video="c1.mp4", span=(0.0, 1.0))
+ clip1.windows = [
+ _Window(start_frame=0, end_frame=10, llm_inputs={"qwen2.5": {"prompt": "p1"}}),
+ ]
+ clip2 = Clip(uuid=uuid4(), source_video="c2.mp4", span=(1.0, 2.0))
+ clip2.windows = [
+ _Window(start_frame=0, end_frame=10, llm_inputs={"qwen2.5": {"prompt": "p2"}}),
+ ]
+ video.clips = [clip1, clip2]
+ return VideoTask(dataset_name="test", data=video)
+
+ def test_hook_builds_script_when_enabled(self):
+ stage = CaptionGenerationStage(model_variant="qwen2.5", build_entity_script=True)
+ stage.model = Mock()
+ stage.model.generate.return_value = ["A red car drives fast.", "The car brakes hard."]
+
+ result = stage.process(self._make_task())
+
+ script = result._metadata[VIDEO_SCRIPT_METADATA_KEY]
+ assert isinstance(script, VideoScript)
+ # The shared "car" entity was reconstructed across the two segments.
+ car = next(e for e in script.entities if e.name == "car")
+ assert car.segment_indices == [0, 1]
+ assert car.recurring is True
+
+ def test_hook_skipped_by_default(self):
+ stage = CaptionGenerationStage(model_variant="qwen2.5")
+ stage.model = Mock()
+ stage.model.generate.return_value = ["A red car drives fast.", "The car brakes hard."]
+
+ result = stage.process(self._make_task())
+
+ assert VIDEO_SCRIPT_METADATA_KEY not in result._metadata
Recommended paper: OmniVideo-100K: A Dataset for Audio-Visual Reasoning through Structured Scripts and Evidence Chains
Confidence: high (Remyx relevance 0.98)
Research interest: Curator
License & code availability
π’ Permissive license β safe to adopt.
Apache-2.0(class:permissive, compat: 1.00, source:github)Why this candidate (selected from the lookback pool)
OmniVideo-100K's Entity-Anchored Video Scripting is a richer, structured captioning method (summary + entity list + segment-wise audio-visual descriptions) that drops in as a new ProcessingStage[VideoTask, VideoTask] after CaptionGenerationStage, exactly matching the existing caption-stage I/O contract and reusing the repo's Qwen/Nemotron VLM+LLM vLLM infra. The repo's recent caption-quality-evaluation (NVIDIA-NeMo#1980) and Nemotron OCR SDG (NVIDIA-NeMo#1899) investments confirm active multimodal-SDG work the scripting stage extends. Among the pool it is the only candidate with a clean call site AND permissive (Apache-2.0) reproducible code, while every other candidate is an architecture/training/inference-infra paper with no data-pipeline anchor.
Why this paper is interesting for the team
OmniVideo-100K is highly relevant to NeMo Curator's push for 'Complex Multimodal Data Synthesis' and generating high-quality interleaved data. Curator is actively building pipelines for video caption quality evaluation and Nemotron VLM support. This paper's approach of 'Entity-Anchored Video Scripting' and 'Clue-Guided QA Generation' to transform videos into structured scripts with summaries, entity lists, and segment-wise audio-visual descriptions directly addresses the challenge of maintaining cross-segment referential consistency and deep cross-modal reasoning. This structured approach to multimodal data annotation and synthesis offers a powerful method for Curator to enrich its video and audio datasets beyond simple event descriptions, supporting more sophisticated downstream multimodal model training.
Suggested experiment
Take a sample of existing video datasets processed by NeMo Curator. Apply the 'Entity-Anchored Video Scripting' mechanism to generate structured scripts, focusing on maintaining cross-segment entity consistency. Integrate these enriched scripts with Curator's current video captioning output and evaluate if this structured representation improves the quality of generated captions or the performance of a small, downstream VLM on video QA tasks, compared to unstructured captions.
Proposed implementation
The coding agent wrote a working draft before the downgrade gate fired. Apply locally with
git applyafter saving the block below.Diff (526 lines)
Why the orchestrator opened an Issue instead of a PR
Diff Risk Score 0.97 exceeds the auto-land threshold (0.80)
A calibrated static-diff risk score placed this change in the high band, where RADAR mandates human review over auto-landing. The implementation is attached for a maintainer to land manually.
Diff Risk Score: 0.97 / high band (auto-land threshold 0.80)
Diff features scored
What else Outrider considered this run
7 other candidate(s) considered and rejected
2606.14591v1β AudioDER: A Deduplication-Enhanced Reasoning Dataset for Post-Training Large Audio-Language Models2606.13233v1β ReSET: Accurate Latency-Critical NVFP4 Reasoning via Step-Aware Temperature Scaling2606.13241v1β Brick: Spatial Capability Routing for the Mixture-of-Models (MoM) Paradigm2606.14672v1β Towards Direct Latent-Space Synthesis for Parallel Branches in LLM-Agent Workflows2606.13392v1β MiniMax Sparse Attention2606.10706v1β Unifying Data, Memory, and Compute Efficiency in LLM training: A Survey2606.12195v1β InternVideo3: Agentify Foundation Models with Multimodal Contextual ReasoningOpened by the Remyx Recommendation orchestrator. The diff's calibrated risk crossed the auto-land threshold β routed to Issue for human review per RADAR's risk-aware policy.
Reopen this Issue if you want Outrider to revisit this paper later. While it stays closed, the orchestrator will not re-recommend the same paper.