Prompt48 commited on
Commit
46fcf6a
·
verified ·
1 Parent(s): 958ab38

Upload edit\Qwen3-TTS-test\.venv\Lib\site-packages\transformers\models\grounding_dino\image_processing_grounding_dino_fast.py with huggingface_hub

Browse files
edit//Qwen3-TTS-test//.venv//Lib//site-packages//transformers//models//grounding_dino//image_processing_grounding_dino_fast.py ADDED
@@ -0,0 +1,776 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # 🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨
2
+ # This file was automatically generated from src/transformers/models/grounding_dino/modular_grounding_dino.py.
3
+ # Do NOT edit this file manually as any edits will be overwritten by the generation of
4
+ # the file from the modular. If any change should be done, please apply the change to the
5
+ # modular_grounding_dino.py file directly. One of our CI enforces this.
6
+ # 🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨🚨
7
+ import pathlib
8
+ from typing import TYPE_CHECKING, Any, Optional, Union
9
+
10
+ import torch
11
+ from torchvision.io import read_image
12
+ from torchvision.transforms.v2 import functional as F
13
+
14
+ from ...image_processing_utils import BatchFeature, get_size_dict
15
+ from ...image_processing_utils_fast import (
16
+ BaseImageProcessorFast,
17
+ DefaultFastImageProcessorKwargs,
18
+ SizeDict,
19
+ get_image_size_for_max_height_width,
20
+ get_max_height_width,
21
+ safe_squeeze,
22
+ )
23
+ from ...image_transforms import center_to_corners_format, corners_to_center_format
24
+ from ...image_utils import (
25
+ IMAGENET_DEFAULT_MEAN,
26
+ IMAGENET_DEFAULT_STD,
27
+ AnnotationFormat,
28
+ AnnotationType,
29
+ ChannelDimension,
30
+ ImageInput,
31
+ PILImageResampling,
32
+ get_image_size,
33
+ validate_annotations,
34
+ )
35
+ from ...processing_utils import Unpack
36
+ from ...utils import TensorType, auto_docstring, logging
37
+ from ...utils.import_utils import requires
38
+ from .image_processing_grounding_dino import get_size_with_aspect_ratio
39
+
40
+
41
+ if TYPE_CHECKING:
42
+ from .modeling_grounding_dino import GroundingDinoObjectDetectionOutput
43
+
44
+
45
+ logger = logging.get_logger(__name__)
46
+
47
+
48
+ class GroundingDinoFastImageProcessorKwargs(DefaultFastImageProcessorKwargs):
49
+ r"""
50
+ format (`str`, *optional*, defaults to `AnnotationFormat.COCO_DETECTION`):
51
+ Data format of the annotations. One of "coco_detection" or "coco_panoptic".
52
+ do_convert_annotations (`bool`, *optional*, defaults to `True`):
53
+ Controls whether to convert the annotations to the format expected by the GROUNDING_DINO model. Converts the
54
+ bounding boxes to the format `(center_x, center_y, width, height)` and in the range `[0, 1]`.
55
+ Can be overridden by the `do_convert_annotations` parameter in the `preprocess` method.
56
+ return_segmentation_masks (`bool`, *optional*, defaults to `False`):
57
+ Whether to return segmentation masks.
58
+ """
59
+
60
+ format: Optional[Union[str, AnnotationFormat]]
61
+ do_convert_annotations: Optional[bool]
62
+ return_segmentation_masks: Optional[bool]
63
+
64
+
65
+ SUPPORTED_ANNOTATION_FORMATS = (AnnotationFormat.COCO_DETECTION, AnnotationFormat.COCO_PANOPTIC)
66
+
67
+
68
+ # inspired by https://github.com/facebookresearch/grounding_dino/blob/master/datasets/coco.py#L33
69
+ def convert_coco_poly_to_mask(segmentations, height: int, width: int, device: torch.device) -> torch.Tensor:
70
+ """
71
+ Convert a COCO polygon annotation to a mask.
72
+
73
+ Args:
74
+ segmentations (`list[list[float]]`):
75
+ List of polygons, each polygon represented by a list of x-y coordinates.
76
+ height (`int`):
77
+ Height of the mask.
78
+ width (`int`):
79
+ Width of the mask.
80
+ """
81
+ try:
82
+ from pycocotools import mask as coco_mask
83
+ except ImportError:
84
+ raise ImportError("Pycocotools is not installed in your environment.")
85
+
86
+ masks = []
87
+ for polygons in segmentations:
88
+ rles = coco_mask.frPyObjects(polygons, height, width)
89
+ mask = coco_mask.decode(rles)
90
+ if len(mask.shape) < 3:
91
+ mask = mask[..., None]
92
+ mask = torch.as_tensor(mask, dtype=torch.uint8, device=device)
93
+ mask = torch.any(mask, axis=2)
94
+ masks.append(mask)
95
+ if masks:
96
+ masks = torch.stack(masks, axis=0)
97
+ else:
98
+ masks = torch.zeros((0, height, width), dtype=torch.uint8, device=device)
99
+
100
+ return masks
101
+
102
+
103
+ # inspired by https://github.com/facebookresearch/grounding_dino/blob/master/datasets/coco.py#L50
104
+ def prepare_coco_detection_annotation(
105
+ image,
106
+ target,
107
+ return_segmentation_masks: bool = False,
108
+ input_data_format: Optional[Union[ChannelDimension, str]] = None,
109
+ ):
110
+ """
111
+ Convert the target in COCO format into the format expected by GROUNDING_DINO.
112
+ """
113
+ image_height, image_width = image.size()[-2:]
114
+
115
+ image_id = target["image_id"]
116
+ image_id = torch.as_tensor([image_id], dtype=torch.int64, device=image.device)
117
+
118
+ # Get all COCO annotations for the given image.
119
+ annotations = target["annotations"]
120
+ classes = []
121
+ area = []
122
+ boxes = []
123
+ keypoints = []
124
+ for obj in annotations:
125
+ if "iscrowd" not in obj or obj["iscrowd"] == 0:
126
+ classes.append(obj["category_id"])
127
+ area.append(obj["area"])
128
+ boxes.append(obj["bbox"])
129
+ if "keypoints" in obj:
130
+ keypoints.append(obj["keypoints"])
131
+
132
+ classes = torch.as_tensor(classes, dtype=torch.int64, device=image.device)
133
+ area = torch.as_tensor(area, dtype=torch.float32, device=image.device)
134
+ iscrowd = torch.zeros_like(classes, dtype=torch.int64, device=image.device)
135
+ # guard against no boxes via resizing
136
+ boxes = torch.as_tensor(boxes, dtype=torch.float32, device=image.device).reshape(-1, 4)
137
+ boxes[:, 2:] += boxes[:, :2]
138
+ boxes[:, 0::2] = boxes[:, 0::2].clip(min=0, max=image_width)
139
+ boxes[:, 1::2] = boxes[:, 1::2].clip(min=0, max=image_height)
140
+
141
+ keep = (boxes[:, 3] > boxes[:, 1]) & (boxes[:, 2] > boxes[:, 0])
142
+
143
+ new_target = {
144
+ "image_id": image_id,
145
+ "class_labels": classes[keep],
146
+ "boxes": boxes[keep],
147
+ "area": area[keep],
148
+ "iscrowd": iscrowd[keep],
149
+ "orig_size": torch.as_tensor([int(image_height), int(image_width)], dtype=torch.int64, device=image.device),
150
+ }
151
+
152
+ if keypoints:
153
+ keypoints = torch.as_tensor(keypoints, dtype=torch.float32, device=image.device)
154
+ # Apply the keep mask here to filter the relevant annotations
155
+ keypoints = keypoints[keep]
156
+ num_keypoints = keypoints.shape[0]
157
+ keypoints = keypoints.reshape((-1, 3)) if num_keypoints else keypoints
158
+ new_target["keypoints"] = keypoints
159
+
160
+ if return_segmentation_masks:
161
+ segmentation_masks = [obj["segmentation"] for obj in annotations]
162
+ masks = convert_coco_poly_to_mask(segmentation_masks, image_height, image_width, device=image.device)
163
+ new_target["masks"] = masks[keep]
164
+
165
+ return new_target
166
+
167
+
168
+ def masks_to_boxes(masks: torch.Tensor) -> torch.Tensor:
169
+ """
170
+ Compute the bounding boxes around the provided panoptic segmentation masks.
171
+
172
+ Args:
173
+ masks: masks in format `[number_masks, height, width]` where N is the number of masks
174
+
175
+ Returns:
176
+ boxes: bounding boxes in format `[number_masks, 4]` in xyxy format
177
+ """
178
+ if masks.numel() == 0:
179
+ return torch.zeros((0, 4), device=masks.device)
180
+
181
+ h, w = masks.shape[-2:]
182
+ y = torch.arange(0, h, dtype=torch.float32, device=masks.device)
183
+ x = torch.arange(0, w, dtype=torch.float32, device=masks.device)
184
+ # see https://github.com/pytorch/pytorch/issues/50276
185
+ y, x = torch.meshgrid(y, x, indexing="ij")
186
+
187
+ x_mask = masks * torch.unsqueeze(x, 0)
188
+ x_max = x_mask.view(x_mask.shape[0], -1).max(-1)[0]
189
+ x_min = (
190
+ torch.where(masks, x.unsqueeze(0), torch.tensor(1e8, device=masks.device)).view(masks.shape[0], -1).min(-1)[0]
191
+ )
192
+
193
+ y_mask = masks * torch.unsqueeze(y, 0)
194
+ y_max = y_mask.view(y_mask.shape[0], -1).max(-1)[0]
195
+ y_min = (
196
+ torch.where(masks, y.unsqueeze(0), torch.tensor(1e8, device=masks.device)).view(masks.shape[0], -1).min(-1)[0]
197
+ )
198
+
199
+ return torch.stack([x_min, y_min, x_max, y_max], 1)
200
+
201
+
202
+ # 2 functions below adapted from https://github.com/cocodataset/panopticapi/blob/master/panopticapi/utils.py
203
+ # Copyright (c) 2018, Alexander Kirillov
204
+ # All rights reserved.
205
+ def rgb_to_id(color):
206
+ """
207
+ Converts RGB color to unique ID.
208
+ """
209
+ if isinstance(color, torch.Tensor) and len(color.shape) == 3:
210
+ if color.dtype == torch.uint8:
211
+ color = color.to(torch.int32)
212
+ return color[:, :, 0] + 256 * color[:, :, 1] + 256 * 256 * color[:, :, 2]
213
+ return int(color[0] + 256 * color[1] + 256 * 256 * color[2])
214
+
215
+
216
+ def prepare_coco_panoptic_annotation(
217
+ image: torch.Tensor,
218
+ target: dict,
219
+ masks_path: Union[str, pathlib.Path],
220
+ return_masks: bool = True,
221
+ input_data_format: Union[ChannelDimension, str] = None,
222
+ ) -> dict:
223
+ """
224
+ Prepare a coco panoptic annotation for GROUNDING_DINO.
225
+ """
226
+ image_height, image_width = get_image_size(image, channel_dim=input_data_format)
227
+ annotation_path = pathlib.Path(masks_path) / target["file_name"]
228
+
229
+ new_target = {}
230
+ new_target["image_id"] = torch.as_tensor(
231
+ [target["image_id"] if "image_id" in target else target["id"]], dtype=torch.int64, device=image.device
232
+ )
233
+ new_target["size"] = torch.as_tensor([image_height, image_width], dtype=torch.int64, device=image.device)
234
+ new_target["orig_size"] = torch.as_tensor([image_height, image_width], dtype=torch.int64, device=image.device)
235
+
236
+ if "segments_info" in target:
237
+ masks = read_image(annotation_path).permute(1, 2, 0).to(dtype=torch.int32, device=image.device)
238
+ masks = rgb_to_id(masks)
239
+
240
+ ids = torch.as_tensor([segment_info["id"] for segment_info in target["segments_info"]], device=image.device)
241
+ masks = masks == ids[:, None, None]
242
+ masks = masks.to(torch.bool)
243
+ if return_masks:
244
+ new_target["masks"] = masks
245
+ new_target["boxes"] = masks_to_boxes(masks)
246
+ new_target["class_labels"] = torch.as_tensor(
247
+ [segment_info["category_id"] for segment_info in target["segments_info"]],
248
+ dtype=torch.int64,
249
+ device=image.device,
250
+ )
251
+ new_target["iscrowd"] = torch.as_tensor(
252
+ [segment_info["iscrowd"] for segment_info in target["segments_info"]],
253
+ dtype=torch.int64,
254
+ device=image.device,
255
+ )
256
+ new_target["area"] = torch.as_tensor(
257
+ [segment_info["area"] for segment_info in target["segments_info"]],
258
+ dtype=torch.float32,
259
+ device=image.device,
260
+ )
261
+
262
+ return new_target
263
+
264
+
265
+ def _scale_boxes(boxes, target_sizes):
266
+ """
267
+ Scale batch of bounding boxes to the target sizes.
268
+
269
+ Args:
270
+ boxes (`torch.Tensor` of shape `(batch_size, num_boxes, 4)`):
271
+ Bounding boxes to scale. Each box is expected to be in (x1, y1, x2, y2) format.
272
+ target_sizes (`list[tuple[int, int]]` or `torch.Tensor` of shape `(batch_size, 2)`):
273
+ Target sizes to scale the boxes to. Each target size is expected to be in (height, width) format.
274
+
275
+ Returns:
276
+ `torch.Tensor` of shape `(batch_size, num_boxes, 4)`: Scaled bounding boxes.
277
+ """
278
+
279
+ if isinstance(target_sizes, (list, tuple)):
280
+ image_height = torch.tensor([i[0] for i in target_sizes])
281
+ image_width = torch.tensor([i[1] for i in target_sizes])
282
+ elif isinstance(target_sizes, torch.Tensor):
283
+ image_height, image_width = target_sizes.unbind(1)
284
+ else:
285
+ raise TypeError("`target_sizes` must be a list, tuple or torch.Tensor")
286
+
287
+ scale_factor = torch.stack([image_width, image_height, image_width, image_height], dim=1)
288
+ scale_factor = scale_factor.unsqueeze(1).to(boxes.device)
289
+ boxes = boxes * scale_factor
290
+ return boxes
291
+
292
+
293
+ @auto_docstring
294
+ @requires(backends=("torchvision", "torch"))
295
+ class GroundingDinoImageProcessorFast(BaseImageProcessorFast):
296
+ resample = PILImageResampling.BILINEAR
297
+ image_mean = IMAGENET_DEFAULT_MEAN
298
+ image_std = IMAGENET_DEFAULT_STD
299
+ format = AnnotationFormat.COCO_DETECTION
300
+ do_resize = True
301
+ do_rescale = True
302
+ do_normalize = True
303
+ do_pad = True
304
+ size = {"shortest_edge": 800, "longest_edge": 1333}
305
+ default_to_square = False
306
+ model_input_names = ["pixel_values", "pixel_mask"]
307
+ valid_kwargs = GroundingDinoFastImageProcessorKwargs
308
+
309
+ def __init__(self, **kwargs: Unpack[GroundingDinoFastImageProcessorKwargs]) -> None:
310
+ if "pad_and_return_pixel_mask" in kwargs:
311
+ kwargs["do_pad"] = kwargs.pop("pad_and_return_pixel_mask")
312
+
313
+ size = kwargs.pop("size", None)
314
+ if "max_size" in kwargs:
315
+ logger.warning_once(
316
+ "The `max_size` parameter is deprecated and will be removed in v4.26. "
317
+ "Please specify in `size['longest_edge'] instead`.",
318
+ )
319
+ max_size = kwargs.pop("max_size")
320
+ else:
321
+ max_size = None if size is None else 1333
322
+
323
+ size = size if size is not None else {"shortest_edge": 800, "longest_edge": 1333}
324
+ self.size = get_size_dict(size, max_size=max_size, default_to_square=False)
325
+
326
+ # Backwards compatibility
327
+ do_convert_annotations = kwargs.get("do_convert_annotations")
328
+ do_normalize = kwargs.get("do_normalize")
329
+ if do_convert_annotations is None and getattr(self, "do_convert_annotations", None) is None:
330
+ self.do_convert_annotations = do_normalize if do_normalize is not None else self.do_normalize
331
+
332
+ super().__init__(**kwargs)
333
+
334
+ @classmethod
335
+ def from_dict(cls, image_processor_dict: dict[str, Any], **kwargs):
336
+ """
337
+ Overrides the `from_dict` method from the base class to make sure parameters are updated if image processor is
338
+ created using from_dict and kwargs e.g. `GroundingDinoImageProcessorFast.from_pretrained(checkpoint, size=600,
339
+ max_size=800)`
340
+ """
341
+ image_processor_dict = image_processor_dict.copy()
342
+ if "max_size" in kwargs:
343
+ image_processor_dict["max_size"] = kwargs.pop("max_size")
344
+ if "pad_and_return_pixel_mask" in kwargs:
345
+ image_processor_dict["pad_and_return_pixel_mask"] = kwargs.pop("pad_and_return_pixel_mask")
346
+ return super().from_dict(image_processor_dict, **kwargs)
347
+
348
+ def prepare_annotation(
349
+ self,
350
+ image: torch.Tensor,
351
+ target: dict,
352
+ format: Optional[AnnotationFormat] = None,
353
+ return_segmentation_masks: Optional[bool] = None,
354
+ masks_path: Optional[Union[str, pathlib.Path]] = None,
355
+ input_data_format: Optional[Union[str, ChannelDimension]] = None,
356
+ ) -> dict:
357
+ """
358
+ Prepare an annotation for feeding into GROUNDING_DINO model.
359
+ """
360
+ format = format if format is not None else self.format
361
+
362
+ if format == AnnotationFormat.COCO_DETECTION:
363
+ return_segmentation_masks = False if return_segmentation_masks is None else return_segmentation_masks
364
+ target = prepare_coco_detection_annotation(
365
+ image, target, return_segmentation_masks, input_data_format=input_data_format
366
+ )
367
+ elif format == AnnotationFormat.COCO_PANOPTIC:
368
+ return_segmentation_masks = True if return_segmentation_masks is None else return_segmentation_masks
369
+ target = prepare_coco_panoptic_annotation(
370
+ image,
371
+ target,
372
+ masks_path=masks_path,
373
+ return_masks=return_segmentation_masks,
374
+ input_data_format=input_data_format,
375
+ )
376
+ else:
377
+ raise ValueError(f"Format {format} is not supported.")
378
+ return target
379
+
380
+ def resize(
381
+ self,
382
+ image: torch.Tensor,
383
+ size: SizeDict,
384
+ interpolation: Optional["F.InterpolationMode"] = None,
385
+ **kwargs,
386
+ ) -> torch.Tensor:
387
+ """
388
+ Resize the image to the given size. Size can be `min_size` (scalar) or `(height, width)` tuple. If size is an
389
+ int, smaller edge of the image will be matched to this number.
390
+
391
+ Args:
392
+ image (`torch.Tensor`):
393
+ Image to resize.
394
+ size (`SizeDict`):
395
+ Size of the image's `(height, width)` dimensions after resizing. Available options are:
396
+ - `{"height": int, "width": int}`: The image will be resized to the exact size `(height, width)`.
397
+ Do NOT keep the aspect ratio.
398
+ - `{"shortest_edge": int, "longest_edge": int}`: The image will be resized to a maximum size respecting
399
+ the aspect ratio and keeping the shortest edge less or equal to `shortest_edge` and the longest edge
400
+ less or equal to `longest_edge`.
401
+ - `{"max_height": int, "max_width": int}`: The image will be resized to the maximum size respecting the
402
+ aspect ratio and keeping the height less or equal to `max_height` and the width less or equal to
403
+ `max_width`.
404
+ interpolation (`InterpolationMode`, *optional*, defaults to `InterpolationMode.BILINEAR`):
405
+ Resampling filter to use if resizing the image.
406
+ """
407
+ interpolation = interpolation if interpolation is not None else F.InterpolationMode.BILINEAR
408
+ if size.shortest_edge and size.longest_edge:
409
+ # Resize the image so that the shortest edge or the longest edge is of the given size
410
+ # while maintaining the aspect ratio of the original image.
411
+ new_size = get_size_with_aspect_ratio(
412
+ image.size()[-2:],
413
+ size["shortest_edge"],
414
+ size["longest_edge"],
415
+ )
416
+ elif size.max_height and size.max_width:
417
+ new_size = get_image_size_for_max_height_width(image.size()[-2:], size["max_height"], size["max_width"])
418
+ elif size.height and size.width:
419
+ new_size = (size["height"], size["width"])
420
+ else:
421
+ raise ValueError(
422
+ "Size must contain 'height' and 'width' keys or 'shortest_edge' and 'longest_edge' keys. Got"
423
+ f" {size.keys()}."
424
+ )
425
+
426
+ image = F.resize(
427
+ image,
428
+ size=new_size,
429
+ interpolation=interpolation,
430
+ **kwargs,
431
+ )
432
+ return image
433
+
434
+ def resize_annotation(
435
+ self,
436
+ annotation: dict[str, Any],
437
+ orig_size: tuple[int, int],
438
+ target_size: tuple[int, int],
439
+ threshold: float = 0.5,
440
+ interpolation: Optional["F.InterpolationMode"] = None,
441
+ ):
442
+ """
443
+ Resizes an annotation to a target size.
444
+
445
+ Args:
446
+ annotation (`dict[str, Any]`):
447
+ The annotation dictionary.
448
+ orig_size (`tuple[int, int]`):
449
+ The original size of the input image.
450
+ target_size (`tuple[int, int]`):
451
+ The target size of the image, as returned by the preprocessing `resize` step.
452
+ threshold (`float`, *optional*, defaults to 0.5):
453
+ The threshold used to binarize the segmentation masks.
454
+ resample (`InterpolationMode`, defaults to `F.InterpolationMode.NEAREST_EXACT`):
455
+ The resampling filter to use when resizing the masks.
456
+ """
457
+ interpolation = interpolation if interpolation is not None else F.InterpolationMode.NEAREST_EXACT
458
+ ratio_height, ratio_width = [target / orig for target, orig in zip(target_size, orig_size)]
459
+
460
+ new_annotation = {}
461
+ new_annotation["size"] = target_size
462
+
463
+ for key, value in annotation.items():
464
+ if key == "boxes":
465
+ boxes = value
466
+ scaled_boxes = boxes * torch.as_tensor(
467
+ [ratio_width, ratio_height, ratio_width, ratio_height], dtype=torch.float32, device=boxes.device
468
+ )
469
+ new_annotation["boxes"] = scaled_boxes
470
+ elif key == "area":
471
+ area = value
472
+ scaled_area = area * (ratio_width * ratio_height)
473
+ new_annotation["area"] = scaled_area
474
+ elif key == "masks":
475
+ masks = value[:, None]
476
+ masks = [F.resize(mask, target_size, interpolation=interpolation) for mask in masks]
477
+ masks = torch.stack(masks).to(torch.float32)
478
+ masks = masks[:, 0] > threshold
479
+ new_annotation["masks"] = masks
480
+ elif key == "size":
481
+ new_annotation["size"] = target_size
482
+ else:
483
+ new_annotation[key] = value
484
+
485
+ return new_annotation
486
+
487
+ def normalize_annotation(self, annotation: dict, image_size: tuple[int, int]) -> dict:
488
+ image_height, image_width = image_size
489
+ norm_annotation = {}
490
+ for key, value in annotation.items():
491
+ if key == "boxes":
492
+ boxes = value
493
+ boxes = corners_to_center_format(boxes)
494
+ boxes /= torch.as_tensor(
495
+ [image_width, image_height, image_width, image_height], dtype=torch.float32, device=boxes.device
496
+ )
497
+ norm_annotation[key] = boxes
498
+ else:
499
+ norm_annotation[key] = value
500
+ return norm_annotation
501
+
502
+ def _update_annotation_for_padded_image(
503
+ self,
504
+ annotation: dict,
505
+ input_image_size: tuple[int, int],
506
+ output_image_size: tuple[int, int],
507
+ padding,
508
+ update_bboxes,
509
+ ) -> dict:
510
+ """
511
+ Update the annotation for a padded image.
512
+ """
513
+ new_annotation = {}
514
+ new_annotation["size"] = output_image_size
515
+ ratio_height, ratio_width = (input / output for output, input in zip(output_image_size, input_image_size))
516
+
517
+ for key, value in annotation.items():
518
+ if key == "masks":
519
+ masks = value
520
+ masks = F.pad(
521
+ masks,
522
+ padding,
523
+ fill=0,
524
+ )
525
+ masks = safe_squeeze(masks, 1)
526
+ new_annotation["masks"] = masks
527
+ elif key == "boxes" and update_bboxes:
528
+ boxes = value
529
+ boxes *= torch.as_tensor([ratio_width, ratio_height, ratio_width, ratio_height], device=boxes.device)
530
+ new_annotation["boxes"] = boxes
531
+ elif key == "size":
532
+ new_annotation["size"] = output_image_size
533
+ else:
534
+ new_annotation[key] = value
535
+ return new_annotation
536
+
537
+ def pad(
538
+ self,
539
+ image: torch.Tensor,
540
+ padded_size: tuple[int, int],
541
+ annotation: Optional[dict[str, Any]] = None,
542
+ update_bboxes: bool = True,
543
+ fill: int = 0,
544
+ ):
545
+ original_size = image.size()[-2:]
546
+ padding_bottom = padded_size[0] - original_size[0]
547
+ padding_right = padded_size[1] - original_size[1]
548
+ if padding_bottom < 0 or padding_right < 0:
549
+ raise ValueError(
550
+ f"Padding dimensions are negative. Please make sure that the padded size is larger than the "
551
+ f"original size. Got padded size: {padded_size}, original size: {original_size}."
552
+ )
553
+ if original_size != padded_size:
554
+ padding = [0, 0, padding_right, padding_bottom]
555
+ image = F.pad(image, padding, fill=fill)
556
+ if annotation is not None:
557
+ annotation = self._update_annotation_for_padded_image(
558
+ annotation, original_size, padded_size, padding, update_bboxes
559
+ )
560
+
561
+ # Make a pixel mask for the image, where 1 indicates a valid pixel and 0 indicates padding.
562
+ pixel_mask = torch.zeros(padded_size, dtype=torch.int64, device=image.device)
563
+ pixel_mask[: original_size[0], : original_size[1]] = 1
564
+
565
+ return image, pixel_mask, annotation
566
+
567
+ @auto_docstring
568
+ def preprocess(
569
+ self,
570
+ images: ImageInput,
571
+ annotations: Optional[Union[AnnotationType, list[AnnotationType]]] = None,
572
+ masks_path: Optional[Union[str, pathlib.Path]] = None,
573
+ **kwargs: Unpack[GroundingDinoFastImageProcessorKwargs],
574
+ ) -> BatchFeature:
575
+ r"""
576
+ annotations (`AnnotationType` or `list[AnnotationType]`, *optional*):
577
+ List of annotations associated with the image or batch of images. If annotation is for object
578
+ detection, the annotations should be a dictionary with the following keys:
579
+ - "image_id" (`int`): The image id.
580
+ - "annotations" (`list[Dict]`): List of annotations for an image. Each annotation should be a
581
+ dictionary. An image can have no annotations, in which case the list should be empty.
582
+ If annotation is for segmentation, the annotations should be a dictionary with the following keys:
583
+ - "image_id" (`int`): The image id.
584
+ - "segments_info" (`list[Dict]`): List of segments for an image. Each segment should be a dictionary.
585
+ An image can have no segments, in which case the list should be empty.
586
+ - "file_name" (`str`): The file name of the image.
587
+ masks_path (`str` or `pathlib.Path`, *optional*):
588
+ Path to the directory containing the segmentation masks.
589
+ """
590
+ if "pad_and_return_pixel_mask" in kwargs:
591
+ kwargs["do_pad"] = kwargs.pop("pad_and_return_pixel_mask")
592
+ logger.warning_once(
593
+ "The `pad_and_return_pixel_mask` argument is deprecated and will be removed in a future version, "
594
+ "use `do_pad` instead."
595
+ )
596
+
597
+ if "max_size" in kwargs:
598
+ logger.warning_once(
599
+ "The `max_size` argument is deprecated and will be removed in a future version, use"
600
+ " `size['longest_edge']` instead."
601
+ )
602
+ kwargs["size"] = kwargs.pop("max_size")
603
+
604
+ return super().preprocess(images, annotations, masks_path, **kwargs)
605
+
606
+ def _preprocess(
607
+ self,
608
+ images: list["torch.Tensor"],
609
+ annotations: Optional[Union[AnnotationType, list[AnnotationType]]],
610
+ masks_path: Optional[Union[str, pathlib.Path]],
611
+ return_segmentation_masks: bool,
612
+ do_resize: bool,
613
+ size: SizeDict,
614
+ interpolation: Optional["F.InterpolationMode"],
615
+ do_rescale: bool,
616
+ rescale_factor: float,
617
+ do_normalize: bool,
618
+ do_convert_annotations: bool,
619
+ image_mean: Optional[Union[float, list[float]]],
620
+ image_std: Optional[Union[float, list[float]]],
621
+ do_pad: bool,
622
+ pad_size: Optional[SizeDict],
623
+ format: Optional[Union[str, AnnotationFormat]],
624
+ return_tensors: Optional[Union[str, TensorType]],
625
+ **kwargs,
626
+ ) -> BatchFeature:
627
+ """
628
+ Preprocess an image or a batch of images so that it can be used by the model.
629
+ """
630
+ if annotations is not None and isinstance(annotations, dict):
631
+ annotations = [annotations]
632
+
633
+ if annotations is not None and len(images) != len(annotations):
634
+ raise ValueError(
635
+ f"The number of images ({len(images)}) and annotations ({len(annotations)}) do not match."
636
+ )
637
+
638
+ format = AnnotationFormat(format)
639
+ if annotations is not None:
640
+ validate_annotations(format, SUPPORTED_ANNOTATION_FORMATS, annotations)
641
+
642
+ if (
643
+ masks_path is not None
644
+ and format == AnnotationFormat.COCO_PANOPTIC
645
+ and not isinstance(masks_path, (pathlib.Path, str))
646
+ ):
647
+ raise ValueError(
648
+ "The path to the directory containing the mask PNG files should be provided as a"
649
+ f" `pathlib.Path` or string object, but is {type(masks_path)} instead."
650
+ )
651
+
652
+ data = {}
653
+
654
+ processed_images = []
655
+ processed_annotations = []
656
+ pixel_masks = [] # Initialize pixel_masks here
657
+ for image, annotation in zip(images, annotations if annotations is not None else [None] * len(images)):
658
+ # prepare (COCO annotations as a list of Dict -> GROUNDING_DINO target as a single Dict per image)
659
+ if annotations is not None:
660
+ annotation = self.prepare_annotation(
661
+ image,
662
+ annotation,
663
+ format,
664
+ return_segmentation_masks=return_segmentation_masks,
665
+ masks_path=masks_path,
666
+ input_data_format=ChannelDimension.FIRST,
667
+ )
668
+
669
+ if do_resize:
670
+ resized_image = self.resize(image, size=size, interpolation=interpolation)
671
+ if annotations is not None:
672
+ annotation = self.resize_annotation(
673
+ annotation,
674
+ orig_size=image.size()[-2:],
675
+ target_size=resized_image.size()[-2:],
676
+ )
677
+ image = resized_image
678
+ # Fused rescale and normalize
679
+ image = self.rescale_and_normalize(image, do_rescale, rescale_factor, do_normalize, image_mean, image_std)
680
+ if do_convert_annotations and annotations is not None:
681
+ annotation = self.normalize_annotation(annotation, get_image_size(image, ChannelDimension.FIRST))
682
+
683
+ processed_images.append(image)
684
+ processed_annotations.append(annotation)
685
+ images = processed_images
686
+ annotations = processed_annotations if annotations is not None else None
687
+
688
+ if do_pad:
689
+ # depends on all resized image shapes so we need another loop
690
+ if pad_size is not None:
691
+ padded_size = (pad_size.height, pad_size.width)
692
+ else:
693
+ padded_size = get_max_height_width(images)
694
+
695
+ padded_images = []
696
+ padded_annotations = []
697
+ for image, annotation in zip(images, annotations if annotations is not None else [None] * len(images)):
698
+ # Pads images and returns their mask: {'pixel_values': ..., 'pixel_mask': ...}
699
+ if padded_size == image.size()[-2:]:
700
+ padded_images.append(image)
701
+ pixel_masks.append(torch.ones(padded_size, dtype=torch.int64, device=image.device))
702
+ padded_annotations.append(annotation)
703
+ continue
704
+ image, pixel_mask, annotation = self.pad(
705
+ image, padded_size, annotation=annotation, update_bboxes=do_convert_annotations
706
+ )
707
+ padded_images.append(image)
708
+ padded_annotations.append(annotation)
709
+ pixel_masks.append(pixel_mask)
710
+ images = padded_images
711
+ annotations = padded_annotations if annotations is not None else None
712
+ data.update({"pixel_mask": torch.stack(pixel_masks, dim=0)})
713
+
714
+ data.update({"pixel_values": torch.stack(images, dim=0)})
715
+ encoded_inputs = BatchFeature(data, tensor_type=return_tensors)
716
+ if annotations is not None:
717
+ encoded_inputs["labels"] = [
718
+ BatchFeature(annotation, tensor_type=return_tensors) for annotation in annotations
719
+ ]
720
+ return encoded_inputs
721
+
722
+ def post_process_object_detection(
723
+ self,
724
+ outputs: "GroundingDinoObjectDetectionOutput",
725
+ threshold: float = 0.1,
726
+ target_sizes: Optional[Union[TensorType, list[tuple]]] = None,
727
+ ):
728
+ """
729
+ Converts the raw output of [`GroundingDinoForObjectDetection`] into final bounding boxes in (top_left_x, top_left_y,
730
+ bottom_right_x, bottom_right_y) format.
731
+
732
+ Args:
733
+ outputs ([`GroundingDinoObjectDetectionOutput`]):
734
+ Raw outputs of the model.
735
+ threshold (`float`, *optional*, defaults to 0.1):
736
+ Score threshold to keep object detection predictions.
737
+ target_sizes (`torch.Tensor` or `list[tuple[int, int]]`, *optional*):
738
+ Tensor of shape `(batch_size, 2)` or list of tuples (`tuple[int, int]`) containing the target size
739
+ `(height, width)` of each image in the batch. If unset, predictions will not be resized.
740
+
741
+ Returns:
742
+ `list[Dict]`: A list of dictionaries, each dictionary containing the following keys:
743
+ - "scores": The confidence scores for each predicted box on the image.
744
+ - "labels": Indexes of the classes predicted by the model on the image.
745
+ - "boxes": Image bounding boxes in (top_left_x, top_left_y, bottom_right_x, bottom_right_y) format.
746
+ """
747
+ batch_logits, batch_boxes = outputs.logits, outputs.pred_boxes
748
+ batch_size = len(batch_logits)
749
+
750
+ if target_sizes is not None and len(target_sizes) != batch_size:
751
+ raise ValueError("Make sure that you pass in as many target sizes as images")
752
+
753
+ # batch_logits of shape (batch_size, num_queries, num_classes)
754
+ batch_class_logits = torch.max(batch_logits, dim=-1)
755
+ batch_scores = torch.sigmoid(batch_class_logits.values)
756
+ batch_labels = batch_class_logits.indices
757
+
758
+ # Convert to [x0, y0, x1, y1] format
759
+ batch_boxes = center_to_corners_format(batch_boxes)
760
+
761
+ # Convert from relative [0, 1] to absolute [0, height] coordinates
762
+ if target_sizes is not None:
763
+ batch_boxes = _scale_boxes(batch_boxes, target_sizes)
764
+
765
+ results = []
766
+ for scores, labels, boxes in zip(batch_scores, batch_labels, batch_boxes):
767
+ keep = scores > threshold
768
+ scores = scores[keep]
769
+ labels = labels[keep]
770
+ boxes = boxes[keep]
771
+ results.append({"scores": scores, "labels": labels, "boxes": boxes})
772
+
773
+ return results
774
+
775
+
776
+ __all__ = ["GroundingDinoImageProcessorFast"]