Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
30 changes: 29 additions & 1 deletion maestro/trainer/models/qwen_2_5_vl/detection.py
Original file line number Diff line number Diff line change
@@ -1,6 +1,11 @@
import numpy as np
from qwen_vl_utils import smart_resize

# Qwen2.5-VL's vision encoder uses a 14px patch with a 2x2 spatial merge, so the
# resized side lengths must be multiples of 14 * 2. qwen-vl-utils dropped the default
# for `smart_resize(factor=...)` in 0.0.13; passing it explicitly works on every version.
QWEN_2_5_VL_IMAGE_FACTOR = 28


def detections_to_suffix_formatter(
xyxy: np.ndarray,
Expand All @@ -9,9 +14,32 @@ def detections_to_suffix_formatter(
resolution_wh: tuple[int, int],
min_pixels: int,
max_pixels: int,
image_factor: int = QWEN_2_5_VL_IMAGE_FACTOR,
) -> str:
"""Formats detections as the JSON suffix Qwen2.5-VL is trained to emit.

Boxes are rescaled from the source image resolution to the resolution the
processor will actually feed the model, so the coordinates in the training
target match what the model sees.

Args:
xyxy (np.ndarray): Boxes in `(N, 4)` `xyxy` format, in source-image pixels.
class_id (np.ndarray): Class index per box, shape `(N,)`.
classes (list[str]): Class names, indexed by `class_id`.
resolution_wh (tuple[int, int]): Source image `(width, height)`.
min_pixels (int): Lower bound on the resized pixel count.
max_pixels (int): Upper bound on the resized pixel count.
image_factor (int): Side lengths are rounded to a multiple of this value.
Defaults to the Qwen2.5-VL vision encoder's patch size times its
spatial merge size.

Returns:
str: A fenced JSON block of `bbox_2d` / `label` objects.
"""
image_w, image_h = resolution_wh
input_h, input_w = smart_resize(height=image_h, width=image_w, min_pixels=min_pixels, max_pixels=max_pixels)
input_h, input_w = smart_resize(
height=image_h, width=image_w, factor=image_factor, min_pixels=min_pixels, max_pixels=max_pixels
)

xyxy = xyxy / [image_w, image_h, image_w, image_h]
xyxy = xyxy * [input_w, input_h, input_w, input_h]
Expand Down
Empty file.
101 changes: 101 additions & 0 deletions test/meastro/trainer/models/qwen_2_5_vl/test_detection.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,101 @@
import numpy as np
import pytest

# `qwen_vl_utils` ships with the optional `qwen_2_5_vl` extra. Without this guard the
# import below aborts collection for the whole run, taking unrelated suites with it.
pytest.importorskip("qwen_vl_utils", reason="requires the optional `qwen_2_5_vl` extra")

from maestro.trainer.models.qwen_2_5_vl.detection import (
QWEN_2_5_VL_IMAGE_FACTOR,
detections_to_suffix_formatter,
)
Comment on lines +8 to +11

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Badge Skip qwen tests when the optional extra is absent

In the default test environment this new test is collected unconditionally, but tox.ini only installs pytest plus the package’s normal dependencies, while qwen-vl-utils lives under the optional qwen_2_5_vl extra in pyproject.toml. On a standard tox/dev run without that extra, importing maestro.trainer.models.qwen_2_5_vl.detection fails during collection before any tests can run; gate this with pytest.importorskip("qwen_vl_utils") or install the qwen extra in the test env.

Useful? React with 👍 / 👎.


MIN_PIXELS = 256 * 28 * 28
MAX_PIXELS = 1280 * 28 * 28


@pytest.mark.parametrize(
("xyxy", "class_id", "classes", "resolution_wh", "image_factor", "expected"),
[
# 1. Single box on a 640x480 image. smart_resize gives (h=476, w=644), so the
# box is scaled by 644/640 horizontally and 476/480 vertically.
(
np.array([[10.0, 20.0, 110.0, 120.0]]),
np.array([0]),
["cat", "dog"],
(640, 480),
QWEN_2_5_VL_IMAGE_FACTOR,
'```json\n[\n\t{"bbox_2d": [10, 19, 110, 119], "label": "cat"}\n]\n```',
),
# 2. Two boxes, second class -> both scaled, labels resolved by class_id.
(
np.array([[0.0, 0.0, 640.0, 480.0], [320.0, 240.0, 640.0, 480.0]]),
np.array([0, 1]),
["cat", "dog"],
(640, 480),
QWEN_2_5_VL_IMAGE_FACTOR,
'```json\n[\n\t{"bbox_2d": [0, 0, 644, 476], "label": "cat"},\n'
'\t{"bbox_2d": [322, 238, 644, 476], "label": "dog"}\n]\n```',
),
# 3. Larger source resolution -> (h=756, w=1036).
(
np.array([[0.0, 0.0, 1024.0, 768.0]]),
np.array([0]),
["cat"],
(1024, 768),
QWEN_2_5_VL_IMAGE_FACTOR,
'```json\n[\n\t{"bbox_2d": [0, 0, 1036, 756], "label": "cat"}\n]\n```',
),
# 4. A non-default factor is honoured: 640x480 at factor 56 gives (h=504, w=616).
(
np.array([[0.0, 0.0, 640.0, 480.0]]),
np.array([0]),
["cat"],
(640, 480),
56,
'```json\n[\n\t{"bbox_2d": [0, 0, 616, 504], "label": "cat"}\n]\n```',
),
# 5. No detections -> an empty JSON block rather than an error.
(
np.zeros((0, 4), dtype=np.float32),
np.zeros((0,), dtype=np.int32),
["cat"],
(640, 480),
QWEN_2_5_VL_IMAGE_FACTOR,
"```json\n[\n\n]\n```",
),
],
)
def test_detections_to_suffix_formatter(
xyxy: np.ndarray,
class_id: np.ndarray,
classes: list[str],
resolution_wh: tuple[int, int],
image_factor: int,
expected: str,
) -> None:
result = detections_to_suffix_formatter(
xyxy=xyxy,
class_id=class_id,
classes=classes,
resolution_wh=resolution_wh,
min_pixels=MIN_PIXELS,
max_pixels=MAX_PIXELS,
image_factor=image_factor,
)
assert result == expected


def test_resized_side_lengths_are_multiples_of_the_image_factor() -> None:
"""The whole point of `factor`: Qwen2.5-VL cannot patch a side it cannot divide."""
result = detections_to_suffix_formatter(
xyxy=np.array([[0.0, 0.0, 640.0, 480.0]]),
class_id=np.array([0]),
classes=["cat"],
resolution_wh=(640, 480),
min_pixels=MIN_PIXELS,
max_pixels=MAX_PIXELS,
)
x1, y1, x2, y2 = (int(value) for value in result.split("[")[2].split("]")[0].split(","))
assert x2 % QWEN_2_5_VL_IMAGE_FACTOR == 0
assert y2 % QWEN_2_5_VL_IMAGE_FACTOR == 0