Skip to content

Commit c3bac96

Browse files
authored
feat(video): support video inputs in vLLM responses (#2324)
## Summary Add video-input support to the NeMo Gym Responses API and vLLM model backend for multimodal RL training. - Accept `input_video` and `video_url` content parts. - Convert Responses API video inputs into vLLM-compatible chat content. - Reject video parts that do not contain a valid URL. - Prefer prompt and generation token IDs returned by the generation request. - Preserve multimodal processor arguments when falling back to the tokenize endpoint. - Validate that generation token IDs and log probabilities have matching lengths. - Propagate token IDs, log probabilities, and optional training metadata through Gym responses. - Ensure component processes load their owning Gym checkout instead of a stale container installation. ## Motivation Video GRPO requires Gym to preserve video content and the exact prompt and generation tokenization used by vLLM. Using token IDs produced by a separate or differently configured tokenization request can cause rollout and policy log probabilities to reference different token sequences, resulting in incorrect training data and elevated TMPE. ## Validation - Added coverage for `input_video` and `video_url` conversion. - Added coverage for missing video URLs. - Added coverage for native vLLM prompt and generation token IDs. - Added coverage for the tokenize fallback and multimodal processor arguments. - Added coverage for token-ID and log-probability propagation. - Added coverage ensuring component processes prefer the owning Gym checkout. - Exercised through downstream synchronous and asynchronous NeMo RL video GRPO training. ## Dependencies None. ## Related integrations - NVIDIA-NeMo/Megatron-Bridge#5304 `feat(video): enable canonical Nemotron Omni training for V2 MoE checkpoints` - NVIDIA-NeMo/RL#3500 `feat(video): add Gym support for sync and async GRPO` --------- Signed-off-by: Ehsan Hosseini Asl <ehsan.hosseiniasl@gmail.com>
1 parent 082ebb1 commit c3bac96

3 files changed

Lines changed: 191 additions & 6 deletions

File tree

‎nemo_gym/openai_utils.py‎

Lines changed: 41 additions & 5 deletions
Original file line numberDiff line numberDiff line change
@@ -82,15 +82,13 @@
8282
from openai.types.responses.response_function_call_output_item_list_param import (
8383
ResponseFunctionCallOutputItemListParam,
8484
)
85+
from openai.types.responses.response_input_content_param import ResponseInputContentParam
8586
from openai.types.responses.response_input_item import (
8687
ComputerCallOutput,
8788
LocalShellCallOutput,
8889
McpApprovalResponse,
8990
ResponseCustomToolCallOutput,
9091
)
91-
from openai.types.responses.response_input_param import (
92-
ResponseInputMessageContentListParam,
93-
)
9492
from openai.types.responses.response_output_item import (
9593
ImageGenerationCall,
9694
LocalShellCall,
@@ -224,14 +222,46 @@ class NeMoGymResponseOutputMessage(BaseModel):
224222
type: Literal["message"] = "message"
225223

226224

225+
class NeMoGymInputVideoPart(TypedDict, total=False):
226+
"""Video content accepted by the Responses-compatible Gym API."""
227+
228+
type: Required[Literal["input_video"]]
229+
video_url: Union[str, Dict[str, Any]]
230+
video: Union[str, Dict[str, Any]]
231+
232+
233+
def _validate_input_video_part(value: Any) -> Any:
234+
"""Require one non-empty source without changing the TypedDict representation."""
235+
236+
if not isinstance(value, dict) or value.get("type") != "input_video":
237+
return value
238+
239+
source_keys = [key for key in ("video_url", "video") if key in value]
240+
if len(source_keys) != 1:
241+
raise ValueError("input_video requires exactly one of video_url or video")
242+
243+
source = value[source_keys[0]]
244+
url = source.get("url") if isinstance(source, dict) else source
245+
if not isinstance(url, str) or not url.strip():
246+
raise ValueError(f"input_video.{source_keys[0]} must contain a non-empty URL")
247+
return value
248+
249+
250+
NeMoGymResponseInputContentPart: TypeAlias = Annotated[
251+
Union[ResponseInputContentParam, NeMoGymInputVideoPart],
252+
BeforeValidator(_validate_input_video_part),
253+
]
254+
NeMoGymResponseInputContentList: TypeAlias = List[NeMoGymResponseInputContentPart]
255+
256+
227257
class NeMoGymEasyInputMessage(BaseModel):
228-
content: Union[str, ResponseInputMessageContentListParam]
258+
content: Union[str, NeMoGymResponseInputContentList]
229259
role: Literal["user", "assistant", "system", "developer"]
230260
type: Literal["message"] = "message"
231261

232262

233263
class NeMoGymMessage(BaseModel):
234-
content: ResponseInputMessageContentListParam
264+
content: NeMoGymResponseInputContentList
235265
role: Literal["user", "system", "developer"]
236266
status: Literal["in_progress", "completed", "incomplete"] = "completed"
237267
type: Literal["message"] = "message"
@@ -678,11 +708,17 @@ class NeMoGymChatCompletionContentPartFileParam(ChatCompletionContentPartFilePar
678708
pass
679709

680710

711+
class NeMoGymChatCompletionContentPartVideoUrlParam(TypedDict, total=False):
712+
video_url: Required[Union[str, Dict[str, Any]]]
713+
type: Required[Literal["video_url"]]
714+
715+
681716
NeMoGymChatCompletionContentPartParam = Union[
682717
NeMoGymChatCompletionContentPartTextParam,
683718
NeMoGymChatCompletionContentPartImageParam,
684719
NeMoGymChatCompletionContentPartInputAudioParam,
685720
NeMoGymChatCompletionContentPartFileParam,
721+
NeMoGymChatCompletionContentPartVideoUrlParam,
686722
]
687723

688724

‎nemo_gym/responses_converter.py‎

Lines changed: 12 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -319,10 +319,22 @@ def _format_message(
319319
converted_parts.append({"type": "text", "text": part_param["text"]})
320320
case "input_image":
321321
image_url = part_param.get("image_url", "")
322+
if isinstance(image_url, dict):
323+
image_url = image_url.get("url", "")
324+
if not image_url:
325+
raise ValueError(f"{part_param['type']} requires a non-empty image_url")
322326
detail = part_param.get("detail", "auto")
323327
converted_parts.append(
324328
{"type": "image_url", "image_url": {"url": image_url, "detail": detail}}
325329
)
330+
case "input_video":
331+
source_key = "video_url" if "video_url" in part_param else "video"
332+
video_url = part_param[source_key]
333+
if isinstance(video_url, dict):
334+
video_url = video_url.get("url", "")
335+
if not video_url:
336+
raise ValueError(f"input_video.{source_key} requires a non-empty URL")
337+
converted_parts.append({"type": "video_url", "video_url": {"url": video_url}})
326338
case _:
327339
raise NotImplementedError(f"Unsupported part param type: {part_param['type']}")
328340
content = converted_parts

‎tests/unit_tests/test_responses_converter.py‎

Lines changed: 138 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -244,7 +244,9 @@ def test_responses_to_chat_completion_no_instructions_adds_no_message(converter:
244244
assert [m["role"] for m in params.messages] == ["user"]
245245

246246

247-
def test_responses_to_chat_completion_input_image_part(converter: ResponsesConverter):
247+
def test_responses_to_chat_completion_input_image_part(
248+
converter: ResponsesConverter,
249+
):
248250
params = converter.responses_to_chat_completion_create_params(
249251
NeMoGymResponseCreateParamsNonStreaming(
250252
input=[
@@ -264,6 +266,141 @@ def test_responses_to_chat_completion_input_image_part(converter: ResponsesConve
264266
assert {"type": "image_url", "image_url": {"url": "http://img", "detail": "high"}} in parts
265267

266268

269+
def test_responses_to_chat_completion_empty_image_url_raises(
270+
converter: ResponsesConverter,
271+
):
272+
with pytest.raises(ValueError, match="requires a non-empty image_url"):
273+
converter._format_message(
274+
{
275+
"role": "user",
276+
"content": [{"type": "input_image", "image_url": ""}],
277+
},
278+
ResponsesConverterState(return_token_id_information=False),
279+
)
280+
281+
282+
@pytest.mark.parametrize(
283+
("video_field", "video_url"),
284+
[
285+
("video_url", "file:///videos/example.mp4"),
286+
("video_url", {"url": "https://example.com/video.mp4"}),
287+
("video", "file:///videos/example.mp4"),
288+
],
289+
)
290+
def test_responses_to_chat_completion_video_part(
291+
converter: ResponsesConverter,
292+
video_field: str,
293+
video_url: object,
294+
):
295+
params = converter.responses_to_chat_completion_create_params(
296+
NeMoGymResponseCreateParamsNonStreaming(
297+
input=[
298+
{
299+
"role": "user",
300+
"type": "message",
301+
"content": [
302+
{"type": "input_text", "text": "what happens?"},
303+
{"type": "input_video", video_field: video_url},
304+
],
305+
}
306+
]
307+
)
308+
)
309+
310+
expected_url = video_url["url"] if isinstance(video_url, dict) else video_url
311+
assert params.messages[0]["content"] == [
312+
{"type": "text", "text": "what happens?"},
313+
{"type": "video_url", "video_url": {"url": expected_url}},
314+
]
315+
316+
317+
def test_responses_to_chat_completion_empty_video_url_raises(
318+
converter: ResponsesConverter,
319+
):
320+
with pytest.raises(ValueError, match="requires a non-empty URL"):
321+
converter._format_message(
322+
{
323+
"role": "user",
324+
"content": [{"type": "input_video", "video_url": ""}],
325+
},
326+
ResponsesConverterState(return_token_id_information=False),
327+
)
328+
329+
330+
def test_responses_video_schema_preserves_mixed_sdk_items_as_dicts():
331+
request = NeMoGymResponseCreateParamsNonStreaming(
332+
input=[
333+
{
334+
"role": "user",
335+
"type": "message",
336+
"content": [
337+
{"type": "input_file", "file_url": "https://example.com/context.txt"},
338+
{"type": "input_video", "video_url": "https://example.com/video.mp4"},
339+
],
340+
}
341+
]
342+
)
343+
344+
content = request.input[0].content
345+
assert isinstance(content[0], dict)
346+
assert isinstance(content[1], dict)
347+
assert content[0]["type"] == "input_file"
348+
assert content[1]["type"] == "input_video"
349+
350+
351+
@pytest.mark.parametrize(
352+
"video_part",
353+
[
354+
{"type": "input_video"},
355+
{"type": "input_video", "video_url": "", "video": "https://example.com/video.mp4"},
356+
{
357+
"type": "input_video",
358+
"video_url": "https://example.com/a.mp4",
359+
"video": "https://example.com/b.mp4",
360+
},
361+
{"type": "input_video", "video_url": {"url": ""}},
362+
],
363+
)
364+
def test_responses_video_schema_requires_exactly_one_nonempty_source(video_part: dict):
365+
with pytest.raises(ValueError, match="exactly one|non-empty URL"):
366+
NeMoGymResponseCreateParamsNonStreaming(input=[{"role": "user", "type": "message", "content": [video_part]}])
367+
368+
369+
def test_responses_schema_rejects_chat_style_media_aliases():
370+
with pytest.raises(ValueError):
371+
NeMoGymResponseCreateParamsNonStreaming(
372+
input=[
373+
{
374+
"role": "user",
375+
"type": "message",
376+
"content": [{"type": "video_url", "video_url": "https://example.com/video.mp4"}],
377+
}
378+
]
379+
)
380+
381+
382+
def test_chat_schema_accepts_only_canonical_video_url():
383+
request = NeMoGymChatCompletionCreateParamsNonStreaming(
384+
messages=[
385+
{
386+
"role": "user",
387+
"content": [{"type": "video_url", "video_url": {"url": "https://example.com/video.mp4"}}],
388+
}
389+
]
390+
)
391+
assert request.messages[0]["content"][0]["type"] == "video_url"
392+
393+
with pytest.raises(ValueError):
394+
NeMoGymChatCompletionCreateParamsNonStreaming(
395+
messages=[
396+
{
397+
"role": "user",
398+
"content": [{"type": "input_video", "video_url": "https://example.com/video.mp4"}],
399+
}
400+
]
401+
)
402+
403+
267404
def test_responses_to_chat_completion_unsupported_part_raises(converter: ResponsesConverter):
268405
# Exercise the converter directly with an unsupported content part type. A raw
269406
# ResponseCreateParams would reject this at schema-validation time, so we call the

0 commit comments

Comments
 (0)