Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 11 additions & 17 deletions .file_mapping.json
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
{
"_source_commit": "4d9b6cfd3731fdbbca883184937d40a64d2dca52-dirty",
"_dest_commit": "c23e51f2f157ae3e51cfcd86ebfb5464850894f2",
"_generated_at": "2026-09-20T05:50:07Z",
"_source_commit": "461431bae40d4a8ccf27d812fc9d1757fe8d3b96-dirty",
"_dest_commit": "0460be81f16883aa380e716dc6f58c1189481172",
"_generated_at": "2026-09-23T04:47:38Z",
"files": {
"imaginaire/__init__.py": "cosmos_framework/__init__.py",
"imaginaire/attention/__init__.py": "cosmos_framework/model/attention/__init__.py",
Expand Down Expand Up @@ -169,6 +169,7 @@
"imaginaire/utils/one_logger/one_logger_global_vars.py": "cosmos_framework/utils/one_logger/one_logger_global_vars.py",
"imaginaire/utils/one_logger/one_logger_override_utils.py": "cosmos_framework/utils/one_logger/one_logger_override_utils.py",
"imaginaire/utils/one_logger/one_logger_utils.py": "cosmos_framework/utils/one_logger/one_logger_utils.py",
"imaginaire/utils/one_logger/one_logger_utils_test.py": "cosmos_framework/utils/one_logger/one_logger_utils_test.py",
"imaginaire/utils/optim_instantiate.py": "cosmos_framework/utils/optim_instantiate.py",
"imaginaire/utils/profiling.py": "cosmos_framework/utils/profiling.py",
"imaginaire/utils/progress_bar.py": "cosmos_framework/utils/progress_bar.py",
Expand Down Expand Up @@ -292,20 +293,6 @@
"projects/cosmos3/cosmos3/datasets/augmentors/cropping.py": "cosmos_framework/data/generator/augmentors/cropping.py",
"projects/cosmos3/cosmos3/datasets/augmentors/duration_fps_text_timestamps.py": "cosmos_framework/data/generator/augmentors/duration_fps_text_timestamps.py",
"projects/cosmos3/cosmos3/datasets/augmentors/duration_fps_text_timestamps_test.py": "cosmos_framework/data/generator/augmentors/duration_fps_text_timestamps_test.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/__init__.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/__init__.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/augmentor.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/augmentor.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/augmentor_test.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/augmentor_test.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/bench.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/bench.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/codec.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/codec.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/contact_sheet.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/contact_sheet.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/contact_sheet_test.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/contact_sheet_test.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/degrade.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/degrade.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/degrade_test.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/degrade_test.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/diffjpeg.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/diffjpeg.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/kernels.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/kernels.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/ops.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/ops.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/packing_test.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/packing_test.py",
"projects/cosmos3/cosmos3/datasets/augmentors/hr_lr_degradation/profiles.py": "cosmos_framework/data/generator/augmentors/hr_lr_degradation/profiles.py",
"projects/cosmos3/cosmos3/datasets/augmentors/idle_frames_text_info.py": "cosmos_framework/data/generator/augmentors/idle_frames_text_info.py",
"projects/cosmos3/cosmos3/datasets/augmentors/image_editing_transform.py": "cosmos_framework/data/generator/augmentors/image_editing_transform.py",
"projects/cosmos3/cosmos3/datasets/augmentors/image_editing_transform_test.py": "cosmos_framework/data/generator/augmentors/image_editing_transform_test.py",
Expand All @@ -326,7 +313,9 @@
"projects/cosmos3/cosmos3/datasets/augmentors/reasoner/format_hot_fixes.py": "cosmos_framework/data/generator/augmentors/reasoner/format_hot_fixes.py",
"projects/cosmos3/cosmos3/datasets/augmentors/reasoner/nvlm_data_to_conversation.py": "cosmos_framework/data/generator/augmentors/reasoner/nvlm_data_to_conversation.py",
"projects/cosmos3/cosmos3/datasets/augmentors/reasoner/prompt_format.py": "cosmos_framework/data/generator/augmentors/reasoner/prompt_format.py",
"projects/cosmos3/cosmos3/datasets/augmentors/reasoner/prompt_format_test.py": "cosmos_framework/data/generator/augmentors/reasoner/prompt_format_test.py",
"projects/cosmos3/cosmos3/datasets/augmentors/reasoner/shuffle_text_media_order.py": "cosmos_framework/data/generator/augmentors/reasoner/shuffle_text_media_order.py",
"projects/cosmos3/cosmos3/datasets/augmentors/reasoner/source_timestamps_test.py": "cosmos_framework/data/generator/augmentors/reasoner/source_timestamps_test.py",
"projects/cosmos3/cosmos3/datasets/augmentors/reasoner/timestamp.py": "cosmos_framework/data/generator/augmentors/reasoner/timestamp.py",
"projects/cosmos3/cosmos3/datasets/augmentors/reasoner/timestamp_test.py": "cosmos_framework/data/generator/augmentors/reasoner/timestamp_test.py",
"projects/cosmos3/cosmos3/datasets/augmentors/reasoner/timestamp_with_subject_tracking.py": "cosmos_framework/data/generator/augmentors/reasoner/timestamp_with_subject_tracking.py",
Expand Down Expand Up @@ -488,9 +477,11 @@
"projects/cosmos3/cosmos3/models/utils/sr_latent_noise_test.py": "cosmos_framework/model/generator/utils/sr_latent_noise_test.py",
"projects/cosmos3/cosmos3/models/vision_encoder.py": "cosmos_framework/model/generator/vision_encoder.py",
"projects/cosmos3/cosmos3/models/vlm_model.py": "cosmos_framework/model/generator/vlm_model.py",
"projects/cosmos3/cosmos3/processors/VIDEO_TIMESTAMPS.md": "cosmos_framework/data/generator/processors/VIDEO_TIMESTAMPS.md",
"projects/cosmos3/cosmos3/processors/__init__.py": "cosmos_framework/data/generator/processors/__init__.py",
"projects/cosmos3/cosmos3/processors/audio_utils.py": "cosmos_framework/data/generator/processors/audio_utils.py",
"projects/cosmos3/cosmos3/processors/base.py": "cosmos_framework/data/generator/processors/base.py",
"projects/cosmos3/cosmos3/processors/base_test.py": "cosmos_framework/data/generator/processors/base_test.py",
"projects/cosmos3/cosmos3/processors/cosmos3_edge_processing.py": "cosmos_framework/data/generator/processors/cosmos3_edge_processing.py",
"projects/cosmos3/cosmos3/processors/cosmos3_edge_processing_test.py": "cosmos_framework/data/generator/processors/cosmos3_edge_processing_test.py",
"projects/cosmos3/cosmos3/processors/nemotron3densevl_processor.py": "cosmos_framework/data/generator/processors/nemotron3densevl_processor.py",
Expand All @@ -504,6 +495,7 @@
"projects/cosmos3/cosmos3/processors/qwen3vl_nemo_chat_processor.py": "cosmos_framework/data/generator/processors/qwen3vl_nemo_chat_processor.py",
"projects/cosmos3/cosmos3/processors/qwen3vl_nemo_chat_processor_test.py": "cosmos_framework/data/generator/processors/qwen3vl_nemo_chat_processor_test.py",
"projects/cosmos3/cosmos3/processors/qwen3vl_processor.py": "cosmos_framework/data/generator/processors/qwen3vl_processor.py",
"projects/cosmos3/cosmos3/processors/source_video_timing_test.py": "cosmos_framework/data/generator/processors/source_video_timing_test.py",
"projects/cosmos3/cosmos3/scripts/multiview_auto/multiview_collage.py": "cosmos_framework/scripts/multiview_collage.py",
"projects/cosmos3/cosmos3/scripts/multiview_auto/multiview_collage_test.py": "cosmos_framework/scripts/multiview_collage_test.py",
"projects/cosmos3/cosmos3/sequence_packing/__init__.py": "cosmos_framework/data/generator/sequence_packing/__init__.py",
Expand Down Expand Up @@ -576,8 +568,10 @@
"projects/cosmos3/cosmos3/utils/reasoner/pretrained_models_downloader.py": "cosmos_framework/utils/generator/reasoner/pretrained_models_downloader.py",
"projects/cosmos3/cosmos3/utils/reasoner/pretrained_models_downloader_test.py": "cosmos_framework/utils/generator/reasoner/pretrained_models_downloader_test.py",
"projects/cosmos3/cosmos3/utils/reasoner/true_packing.py": "cosmos_framework/utils/generator/reasoner/true_packing.py",
"projects/cosmos3/cosmos3/utils/source_video_timing.py": "cosmos_framework/utils/generator/source_video_timing.py",
"projects/cosmos3/cosmos3/utils/video_frame_sampling.py": "cosmos_framework/utils/generator/video_frame_sampling.py",
"projects/cosmos3/cosmos3/utils/video_preprocess.py": "cosmos_framework/utils/generator/video_preprocess.py",
"projects/cosmos3/cosmos3/utils/video_source_metadata.py": "cosmos_framework/utils/generator/video_source_metadata.py",
"projects/cosmos3/interactive/configs/defaults/flex_attention.py": "cosmos_framework/configs/base/defaults/causal_flex_attention.py",
"projects/cosmos3/interactive/configs/defaults/replay_attention.py": "cosmos_framework/configs/base/defaults/replay_attention.py",
"projects/cosmos3/interactive/models/attention_io_layout.py": "cosmos_framework/model/generator/attention_io_layout.py",
Expand Down
7 changes: 5 additions & 2 deletions cosmos_framework/configs/base/defaults/multiview_attention.py
Original file line number Diff line number Diff line change
Expand Up @@ -78,8 +78,11 @@ def resolve_caption_scope(access: CaptionAccess, *, per_view_captions: bool) ->
# ``decomposed_temporal_window_seconds`` is set: the two streams do not share a frame
# index, but they do share real capture time, which the window compares instead.
#
# Read by the ``flex_*`` backends only. The ``"maskless"`` backend is its own attention pattern
# and does not take a scope -- see ``BackendPreference``.
# Read by every backend, but not the same way. A ``flex_*`` backend expresses the scope as a mask.
# The ``"maskless"`` backend expresses ``"same_view"`` and ``"decomposed"`` as partitions of the
# GEN stream and refuses ``"all_views"``, which is not a partition at all -- so there the scope
# decides whether the cross-instant pass exists rather than describing one attention two ways. See
# ``BackendPreference`` and ``models.mot.multiview_maskless_attention.MASKLESS_ATTENTION_SCOPES``.
AttentionScope = Literal["all_views", "same_view", "decomposed"]

# The scopes of ``AttentionScope`` at runtime, which the annotation itself is not.
Expand Down
4 changes: 4 additions & 0 deletions cosmos_framework/configs/base/reasoner/defaults/augmentors.py
Original file line number Diff line number Diff line change
Expand Up @@ -37,6 +37,7 @@ def create_data_augmentor_config() -> dict[str, Any]:
max_fps_thres=60,
target_fps="${data_setting.qwen_target_fps}", # type: ignore
video_temporal_mode="${data_setting.qwen_video_temporal_mode}",
video_timestamp_mode="${data_setting.video_timestamp_mode}",
max_video_token_length="${data_setting.qwen_max_video_token_length}", # type: ignore
processor=processor,
extract_audio="${model.config.sound_und}",
Expand All @@ -45,6 +46,7 @@ def create_data_augmentor_config() -> dict[str, Any]:
"prompt_format": L(PromptFormat)( # takes text_keys and output "conversation"
input_keys=["texts"],
text_chat_order="${data_setting.text_chat_order}",
strip_thinking_prob="${data_setting.strip_thinking_prob}",
),
"shuffle_text_media_order": L(ShuffleTextMediaOrder)(),
"format_hot_fixes": L(FormatHotFixes)(),
Expand Down Expand Up @@ -130,6 +132,7 @@ def create_data_augmentor_config() -> dict[str, Any]:
custom_system_prompt="${data_setting.custom_system_prompt}",
strip_original_system_prompt="${data_setting.strip_original_system_prompt}",
video_temporal_mode="${data_setting.qwen_video_temporal_mode}",
video_timestamp_mode="${data_setting.video_timestamp_mode}",
text_only=False,
sound_und="${model.config.sound_und}",
audio_encoder_type="${model.config.sound_und_config.audio_encoder_type}",
Expand Down Expand Up @@ -168,6 +171,7 @@ def create_data_augmentor_config() -> dict[str, Any]:

processor = L(build_processor_lazy)(
tokenizer_type="${model.config.policy.backbone.model_name}",
use_native_edge_processor="${data_setting.use_native_edge_processor}",
credentials="${checkpoint.load_from_object_store.credentials}",
bucket="${checkpoint.load_from_object_store.bucket}",
)
14 changes: 14 additions & 0 deletions cosmos_framework/configs/base/reasoner/defaults/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@ class DataSetting:
qwen_max_video_token_length: Maximum video token length.
qwen_target_fps: Target fps for video sampling.
text_chat_order: Order of text items in user messages.
strip_thinking_prob: Per-sample probability of converting thinking data into non-thinking data.
custom_system_prompt: System prompt injected when a conversation has no leading system message.
strip_original_system_prompt: Remove existing system messages before optional custom prompt injection.
distributor_type: "with_replace" (WeightedShardlistBasic) or "no_replace" (NoReplaceShardlistBasic).
Expand All @@ -30,6 +31,11 @@ class DataSetting:
qwen_max_video_token_length: int = 8192
qwen_max_image_token_length: int = 8192
qwen_target_fps: float = 4.0
use_native_edge_processor: bool = False
video_timestamp_mode: str = attrs.field(
default="qwen_index",
validator=attrs.validators.in_({"qwen_index", "legacy_fps", "source_pts"}),
)
qwen_video_temporal_mode: str = attrs.field(
default="native", validator=attrs.validators.in_({"native", "framewise"})
)
Expand All @@ -38,6 +44,14 @@ class DataSetting:
default="text_end",
validator=attrs.validators.in_({"text_end", "text_start", "random"}),
)
strip_thinking_prob: float = attrs.field(
default=0.0,
validator=attrs.validators.and_(
attrs.validators.instance_of((int, float)),
attrs.validators.ge(0.0),
attrs.validators.le(1.0),
),
)
custom_system_prompt: str | None = "You are a helpful assistant."
strip_original_system_prompt: bool = False
temporal_localization_output_format: str = attrs.field(
Expand Down
5 changes: 5 additions & 0 deletions cosmos_framework/data/generator/action/utils/domain_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -54,6 +54,10 @@
# RoboCasa PandaOmron mobile manipulation (10/15/20D raw action per
# ``use_base_action`` / ``base_encoding``); appended above the maximum.
"robocasa": 30,
# embodiment_b nvidia-20260828 ingestion: a new one-shot dataset, distinct from
# "embodiment_b" (domain 9, an earlier unrelated sample drop with its own 30D
# contract).
"embodiment_b_20260828": 32,
}


Expand Down Expand Up @@ -88,6 +92,7 @@
"so101-bimanual-midtrain-conditional": 20,
"geniesim3_g2a": 29,
"geniesim3_g2a_joint": 16,
"embodiment_b_20260828": 50,
# NOTE: ``libero`` (7/10/13 depending on ``rotation_space``), ``hand_pose``
# (variable with ``keypoint_option`` and ``rotation_format``) and ``robocasa``
# (10 arm-only, 15/20 with the mobile base, per ``use_base_action`` /
Expand Down
Loading
Loading