Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 13 additions & 5 deletions .file_mapping.json
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
{
"_source_commit": "845dbfbd3aa3913f3d031aa1f09fb16af8b37aa1-dirty",
"_dest_commit": "cf5d68c00d97ccd2480a2320ed652b92dec63102",
"_generated_at": "2026-10-07T07:59:16Z",
"_source_commit": "c956695327e59d1b759868e8e0a36cca5733cdb3-dirty",
"_dest_commit": "8aca062dd4d93111d8afc4c6e7c7c592a81b81bc",
"_generated_at": "2026-10-08T08:05:55Z",
"files": {
"imaginaire/__init__.py": "cosmos_framework/__init__.py",
"imaginaire/attention/__init__.py": "cosmos_framework/model/attention/__init__.py",
Expand Down Expand Up @@ -199,6 +199,9 @@
"projects/cosmos3/cosmos3/algorithm/loss/flow_matching_test.py": "cosmos_framework/model/generator/algorithm/loss/flow_matching_test.py",
"projects/cosmos3/cosmos3/algorithm/loss/load_balancing.py": "cosmos_framework/model/generator/algorithm/loss/load_balancing.py",
"projects/cosmos3/cosmos3/algorithm/loss/load_balancing_test.py": "cosmos_framework/model/generator/algorithm/loss/load_balancing_test.py",
"projects/cosmos3/cosmos3/algorithm/loss/modality_reduction.py": "cosmos_framework/model/generator/algorithm/loss/modality_reduction.py",
"projects/cosmos3/cosmos3/algorithm/loss/modality_reduction_distributed_test.py": "cosmos_framework/model/generator/algorithm/loss/modality_reduction_distributed_test.py",
"projects/cosmos3/cosmos3/algorithm/loss/modality_reduction_test.py": "cosmos_framework/model/generator/algorithm/loss/modality_reduction_test.py",
"projects/cosmos3/cosmos3/algorithm/loss/time_weight.py": "cosmos_framework/model/generator/algorithm/loss/time_weight.py",
"projects/cosmos3/cosmos3/callbacks/compile_tokenizer.py": "cosmos_framework/callbacks/compile_tokenizer.py",
"projects/cosmos3/cosmos3/callbacks/data_stats.py": "cosmos_framework/callbacks/data_stats.py",
Expand Down Expand Up @@ -489,6 +492,7 @@
"projects/cosmos3/cosmos3/models/reasoner/qwen3_vl_moe/shared_expert_test.py": "cosmos_framework/model/generator/reasoner/qwen3_vl_moe/shared_expert_test.py",
"projects/cosmos3/cosmos3/models/sensor_encoder.py": "cosmos_framework/model/generator/sensor_encoder.py",
"projects/cosmos3/cosmos3/models/utils/__init__.py": "cosmos_framework/model/generator/utils/__init__.py",
"projects/cosmos3/cosmos3/models/utils/batch_normalization.py": "cosmos_framework/model/generator/utils/batch_normalization.py",
"projects/cosmos3/cosmos3/models/utils/data_and_condition.py": "cosmos_framework/model/generator/utils/data_and_condition.py",
"projects/cosmos3/cosmos3/models/utils/data_and_condition_test.py": "cosmos_framework/model/generator/utils/data_and_condition_test.py",
"projects/cosmos3/cosmos3/models/utils/load_balancing_stats.py": "cosmos_framework/model/generator/utils/load_balancing_stats.py",
Expand Down Expand Up @@ -604,7 +608,6 @@
"projects/cosmos3/interactive/configs/defaults/flex_attention.py": "cosmos_framework/configs/base/defaults/causal_flex_attention.py",
"projects/cosmos3/interactive/configs/defaults/replay_attention.py": "cosmos_framework/configs/base/defaults/replay_attention.py",
"projects/cosmos3/interactive/models/attention_io_layout.py": "cosmos_framework/model/generator/attention_io_layout.py",
"projects/cosmos3/interactive/models/joint_transfer_ar.py": "cosmos_framework/model/generator/joint_transfer_ar.py",
"projects/cosmos3/interactive/models/mot/attention.py": "cosmos_framework/model/generator/mot/causal_attention.py",
"projects/cosmos3/interactive/models/mot/context_parallel_test.py": "cosmos_framework/model/generator/mot/causal_context_parallel_test.py",
"projects/cosmos3/interactive/models/mot/cosmos3_vfm_network.py": "cosmos_framework/model/generator/mot/causal_cosmos3_vfm_network.py",
Expand All @@ -619,7 +622,6 @@
"projects/cosmos3/interactive/models/mot/post_saturation/static_compile.py": "cosmos_framework/model/generator/mot/post_saturation/static_compile.py",
"projects/cosmos3/interactive/models/mot/replicated_io.py": "cosmos_framework/model/generator/mot/replicated_io.py",
"projects/cosmos3/interactive/models/mot/three_way_attention_test.py": "cosmos_framework/model/generator/mot/three_way_attention_test.py",
"projects/cosmos3/interactive/models/multiview_transfer_ar.py": "cosmos_framework/model/generator/multiview_transfer_ar.py",
"projects/cosmos3/interactive/models/omni_mot_causal_model.py": "cosmos_framework/model/generator/omni_mot_causal_model.py",
"projects/cosmos3/interactive/models/omni_mot_causal_model_test.py": "cosmos_framework/model/generator/omni_mot_causal_model_test.py",
"projects/cosmos3/interactive/models/rolling_prompt.py": "cosmos_framework/model/generator/rolling_prompt.py",
Expand All @@ -632,8 +634,14 @@
"projects/cosmos3/interactive/models/utils/kv_cache_test.py": "cosmos_framework/model/generator/utils/kv_cache_test.py",
"projects/cosmos3/interactive/models/utils/kv_storage_backend.py": "cosmos_framework/model/generator/utils/kv_storage_backend.py",
"projects/cosmos3/interactive/models/utils/kv_storage_backend_test.py": "cosmos_framework/model/generator/utils/kv_storage_backend_test.py",
"projects/cosmos3/interactive/models/utils/multiview_ar.py": "cosmos_framework/model/generator/utils/multiview_ar.py",
"projects/cosmos3/interactive/models/utils/nvfp4.py": "cosmos_framework/model/generator/utils/nvfp4.py",
"projects/cosmos3/interactive/models/utils/nvfp4_test.py": "cosmos_framework/model/generator/utils/nvfp4_test.py",
"projects/cosmos3/interactive/models/utils/rolling_kv/__init__.py": "cosmos_framework/model/generator/utils/rolling_kv/__init__.py",
"projects/cosmos3/interactive/models/utils/rolling_kv/rolling_prompt.py": "cosmos_framework/model/generator/utils/rolling_kv/rolling_prompt.py",
"projects/cosmos3/interactive/models/utils/rolling_kv/rolling_replay.py": "cosmos_framework/model/generator/utils/rolling_kv/rolling_replay.py",
"projects/cosmos3/interactive/models/utils/rolling_kv/rolling_sink_rope.py": "cosmos_framework/model/generator/utils/rolling_kv/rolling_sink_rope.py",
"projects/cosmos3/interactive/models/utils/rolling_kv/rolling_transfer_ar.py": "cosmos_framework/model/generator/utils/rolling_kv/rolling_transfer_ar.py",
"projects/cosmos3/interactive/sequence_packing.py": "cosmos_framework/data/generator/sequence_packing/autoregressive.py",
"projects/cosmos3/interactive/utils/data_batch.py": "cosmos_framework/utils/generator/data_batch.py",
"projects/cosmos3/tokenizer/lidar_tokenizer/__init__.py": "cosmos_framework/model/generator/tokenizers/lidar/__init__.py",
Expand Down
4 changes: 3 additions & 1 deletion .github/workflows/gpu-tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -361,12 +361,14 @@ jobs:
# and inference2/, are intentionally omitted to keep this single-runner job within
# its time budget. Add an L0 marker when one of those tests should join this gate.
# L1, L2, and unmarked cases remain available for explicit or scheduled runs.
# Non-GPU tests generated from i4 are already covered by i4 CI and are deselected here;
# CF-owned tests plus mapped tests explicitly marked GPU or gpus(N>0) remain in this job.
- name: Unit tests
run: |
export LD_LIBRARY_PATH=
export COSMOS_DOWNLOAD_CACHE_DIR="$RUNNER_WORKSPACE/cosmos_input_cache"
uv run --all-extras --group=cu128-train python -m pytest -v -s \
cosmos_framework/ -o addopts= -m L0 \
cosmos_framework/ -o addopts= -m L0 --skip-mapped-non-gpu-tests \
--deselect cosmos_framework/model/generator/mot/attention_test.py::test_two_way_attention_flex_matches_dense_across_batch_shapes

# Deselected above and run here instead, in a process of its own.
Expand Down
34 changes: 32 additions & 2 deletions conftest.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,13 +7,21 @@
lazy_call._CONVERT_TARGET_TO_STRING = True

import gc
import json
import os
from functools import cache
from pathlib import Path

import pytest

from cosmos_framework.inference.fixtures.args import ALL_LEVELS, ALL_NUM_GPUS, ALLOWED_GPUS_BY_LEVEL, Args, get_args, init_args
from cosmos_framework.inference.fixtures.args import (
ALL_LEVELS,
ALL_NUM_GPUS,
ALLOWED_GPUS_BY_LEVEL,
Args,
get_args,
init_args,
)


@pytest.fixture(scope="module")
Expand All @@ -39,6 +47,12 @@ def _get_available_gpus() -> int:

def pytest_addoption(parser: pytest.Parser):
parser.addoption("--manual", action="store_true", default=False, help="Run manual tests")
parser.addoption(
"--skip-mapped-non-gpu-tests",
action="store_true",
default=False,
help="Skip tests generated from imaginaire4 unless they explicitly require a GPU.",
)
parser.addoption(
"--num-gpus",
default=None,
Expand Down Expand Up @@ -129,13 +143,26 @@ def _parse_gpus_marker(mark: pytest.Mark) -> int:
return required_gpus


@cache
def _mapped_destinations(root_dir: Path) -> frozenset[str]:
"""Return repository-relative files generated from imaginaire4 sources."""
payload = json.loads((root_dir / ".file_mapping.json").read_text())
files = payload.get("files", payload)
if not isinstance(files, dict) or any(not isinstance(path, str) for path in files.values()):
raise TypeError(".file_mapping.json must contain a string-to-string files mapping.")
return frozenset(files.values())


def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item]):
args = get_args()
skip_mapped_non_gpu_tests = bool(config.getoption("--skip-mapped-non-gpu-tests"))
mapped_destinations = _mapped_destinations(config.rootpath) if skip_mapped_non_gpu_tests else frozenset()

for item in items:
manual_mark = _get_marker(item, "manual")
level_mark = _get_marker(item, "level")
gpus_mark = _get_marker(item, "gpus")
gpu_mark = _get_marker(item, "GPU")
try:
level = _parse_level_marker(level_mark) if level_mark else 0
gpus = _parse_gpus_marker(gpus_mark) if gpus_mark else 0
Expand All @@ -159,6 +186,10 @@ def pytest_collection_modifyitems(config: pytest.Config, items: list[pytest.Item
item.add_marker(
pytest.mark.skip(reason=f"test requires {gpus} GPUs, but only {available_gpus} are available")
)
if skip_mapped_non_gpu_tests and gpu_mark is None and gpus == 0:
item_path = Path(str(item.path)).relative_to(config.rootpath).as_posix()
if item_path in mapped_destinations:
item.add_marker(pytest.mark.skip(reason="mapped non-GPU test is covered by imaginaire4 CI"))

# Exclude skipped tests
selected_items = []
Expand Down Expand Up @@ -257,7 +288,6 @@ def init_torch_test():
torch.cuda.empty_cache()



_WHITELIST_ENV_VARS = {
# Both set by sklearn/__init__.py at import time (OpenMP workarounds: allow a
# duplicate OpenMP runtime, and skip the intel-openmp 2019.5 fork bug). Reached
Expand Down
71 changes: 50 additions & 21 deletions cosmos_framework/auxiliary/guardrail/qwen3guard/qwen3guard.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,9 +3,10 @@

import argparse
import re
from dataclasses import dataclass
from typing import Any

import torch
from transformers import AutoModelForCausalLM, AutoTokenizer

from cosmos_framework.auxiliary.guardrail.common.core import ContentSafetyGuardrail, GuardrailRunner
from cosmos_framework.auxiliary.guardrail.qwen3guard.categories import UNSAFE_CATEGORIES
Expand All @@ -15,51 +16,79 @@
UNSAFE = misc.Color.red("UNSAFE")


@dataclass(frozen=True)
class Qwen3GuardClassification:
"""Structured Qwen3Guard generation result."""

label: str
categories: tuple[str, ...]
raw_output: str


class Qwen3Guard(ContentSafetyGuardrail):
offload_model: bool
dtype: torch.dtype
model_id: str
model: Any
tokenizer: Any

def __init__(
self,
offload_model_to_cpu: bool = True,
) -> None:
"""Llama Guard 3 model for text filtering safety check.
"""Load Qwen3Guard for text filtering safety checks.

Args:
checkpoint_dir (str): Path to the checkpoint directory.
offload_model_to_cpu (bool, optional): Whether to offload the model to CPU. Defaults to True.
"""
from transformers import AutoModelForCausalLM, AutoTokenizer

self.offload_model = offload_model_to_cpu
self.dtype = torch.bfloat16

model_id = "Qwen/Qwen3Guard-Gen-0.6B"
self.model_id = "Qwen/Qwen3Guard-Gen-0.6B"

self.model = AutoModelForCausalLM.from_pretrained(model_id)
self.tokenizer = AutoTokenizer.from_pretrained(model_id)
self.model = AutoModelForCausalLM.from_pretrained(self.model_id)
self.tokenizer = AutoTokenizer.from_pretrained(self.model_id)

# Move model to GPU unless offload_model_to_cpu is True
if not offload_model_to_cpu:
self.model = self.model.to("cuda", dtype=self.dtype).eval()
log.debug("Moved llamaGuard3 model to GPU")
log.debug("Moved Qwen3Guard model to GPU")
else:
self.model = self.model.to("cpu", dtype=self.dtype).eval()
log.debug("Moved Qwen3Guard model to CPU")

def extract_label_and_categories(self, prompt):
@staticmethod
def parse_output(content: str) -> Qwen3GuardClassification:
"""Parse Qwen3Guard's generated safety label without collapsing its middle class."""
safe_pattern = r"Safety: (Safe|Unsafe|Controversial)"
category_pattern = r"(" + "|".join(UNSAFE_CATEGORIES.values()) + ")"
category_pattern = r"(" + "|".join(re.escape(value) for value in UNSAFE_CATEGORIES.values()) + ")"
safe_label_match = re.search(safe_pattern, content)
if safe_label_match is None:
raise ValueError(f"Qwen3Guard output did not contain a safety label: {content!r}")
label = safe_label_match.group(1)
categories = tuple(dict.fromkeys(re.findall(category_pattern, content)))
return Qwen3GuardClassification(label=label, categories=categories, raw_output=content)

@torch.inference_mode()
def classify(self, prompt: str) -> Qwen3GuardClassification:
"""Generate and parse one native Safe/Controversial/Unsafe decision."""
messages = [{"role": "user", "content": prompt}]

text = self.tokenizer.apply_chat_template(messages, tokenize=False)
model_inputs = self.tokenizer([text], return_tensors="pt").to(self.model.device)
generated_ids = self.model.generate(**model_inputs, max_new_tokens=128)
output_ids = generated_ids[0][len(model_inputs.input_ids[0]) :].tolist()
model_inputs = self.tokenizer([text], return_tensors="pt").to(self.model.device) # [1, input_tokens]
generated_ids = self.model.generate(**model_inputs, max_new_tokens=128) # [1, input_tokens + output_tokens]
output_ids = generated_ids[0][len(model_inputs.input_ids[0]) :].tolist() # [output_tokens]
content = self.tokenizer.decode(output_ids, skip_special_tokens=True)
return self.parse_output(content)

safe_label_match = re.search(safe_pattern, content)
label = safe_label_match.group(1) if safe_label_match else None
categories = re.findall(category_pattern, content)
if label.lower() == "unsafe":
return False, f"Prompt blocked by Qwen3Guard. Safety: {label}, Categories: {categories}"
else:
return True, ""
def extract_label_and_categories(self, prompt: str) -> tuple[bool, str]:
"""Return the legacy binary result while retaining native classification through classify()."""
result = self.classify(prompt)
if result.label.lower() == "unsafe":
return False, f"Prompt blocked by Qwen3Guard. Safety: {result.label}, Categories: {result.categories}"
return True, ""

def is_safe(self, prompt: str) -> tuple[bool, str]:
"""Check if the input prompt is safe according to the Qwen3Guard model."""
Expand All @@ -70,13 +99,13 @@ def is_safe(self, prompt: str) -> tuple[bool, str]:
return True, "Unexpected error occurred when running Qwen3Guard guardrail."


def parse_args():
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser()
parser.add_argument("--prompt", type=str, required=True, help="Input prompt")
return parser.parse_args()


def main(args):
def main(args: argparse.Namespace) -> None:
qwen3guard = Qwen3Guard()
runner = GuardrailRunner(safety_models=[qwen3guard])
with misc.timer("Qwen3Guard safety check"):
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,9 @@ def encode_image(self, input_img: Image.Image) -> torch.Tensor:
"""Encode an image into a feature vector."""
with torch.no_grad():
inputs = self.processor(images=input_img, return_tensors="pt").to(self.device, dtype=self.dtype)
image_features = self.model.get_image_features(**inputs)
image_features /= image_features.norm(dim=-1, keepdim=True)
model_output = self.model.get_image_features(**inputs)
image_features = getattr(model_output, "pooler_output", model_output) # [B,D]
if not isinstance(image_features, torch.Tensor):
raise TypeError("SigLIP image encoder did not return pooled tensor features")
image_features = torch.nn.functional.normalize(image_features, dim=-1) # [B,D]
return image_features
Loading
Loading