Prompt48 commited on
Commit
8807c75
·
verified ·
1 Parent(s): 8305734

Upload edit\Qwen3-TTS-test\.venv\Lib\site-packages\transformers\models\grounding_dino\configuration_grounding_dino.py with huggingface_hub

Browse files
edit//Qwen3-TTS-test//.venv//Lib//site-packages//transformers//models//grounding_dino//configuration_grounding_dino.py ADDED
@@ -0,0 +1,309 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # coding=utf-8
2
+ # Copyright 2024 The HuggingFace Inc. team. All rights reserved.
3
+ #
4
+ # Licensed under the Apache License, Version 2.0 (the "License");
5
+ # you may not use this file except in compliance with the License.
6
+ # You may obtain a copy of the License at
7
+ #
8
+ # http://www.apache.org/licenses/LICENSE-2.0
9
+ #
10
+ # Unless required by applicable law or agreed to in writing, software
11
+ # distributed under the License is distributed on an "AS IS" BASIS,
12
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
13
+ # See the License for the specific language governing permissions and
14
+ # limitations under the License.
15
+ """Grounding DINO model configuration"""
16
+
17
+ from ...configuration_utils import PretrainedConfig
18
+ from ...utils import logging
19
+ from ...utils.backbone_utils import verify_backbone_config_arguments
20
+ from ..auto import CONFIG_MAPPING
21
+
22
+
23
+ logger = logging.get_logger(__name__)
24
+
25
+
26
+ class GroundingDinoConfig(PretrainedConfig):
27
+ r"""
28
+ This is the configuration class to store the configuration of a [`GroundingDinoModel`]. It is used to instantiate a
29
+ Grounding DINO model according to the specified arguments, defining the model architecture. Instantiating a
30
+ configuration with the defaults will yield a similar configuration to that of the Grounding DINO
31
+ [IDEA-Research/grounding-dino-tiny](https://huggingface.co/IDEA-Research/grounding-dino-tiny) architecture.
32
+
33
+ Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
34
+ documentation from [`PretrainedConfig`] for more information.
35
+
36
+ Args:
37
+ backbone_config (`PretrainedConfig` or `dict`, *optional*, defaults to `ResNetConfig()`):
38
+ The configuration of the backbone model.
39
+ backbone (`str`, *optional*):
40
+ Name of backbone to use when `backbone_config` is `None`. If `use_pretrained_backbone` is `True`, this
41
+ will load the corresponding pretrained weights from the timm or transformers library. If `use_pretrained_backbone`
42
+ is `False`, this loads the backbone's config and uses that to initialize the backbone with random weights.
43
+ use_pretrained_backbone (`bool`, *optional*, defaults to `False`):
44
+ Whether to use pretrained weights for the backbone.
45
+ use_timm_backbone (`bool`, *optional*, defaults to `False`):
46
+ Whether to load `backbone` from the timm library. If `False`, the backbone is loaded from the transformers
47
+ library.
48
+ backbone_kwargs (`dict`, *optional*):
49
+ Keyword arguments to be passed to AutoBackbone when loading from a checkpoint
50
+ e.g. `{'out_indices': (0, 1, 2, 3)}`. Cannot be specified if `backbone_config` is set.
51
+ text_config (`Union[AutoConfig, dict]`, *optional*, defaults to `BertConfig`):
52
+ The config object or dictionary of the text backbone.
53
+ num_queries (`int`, *optional*, defaults to 900):
54
+ Number of object queries, i.e. detection slots. This is the maximal number of objects
55
+ [`GroundingDinoModel`] can detect in a single image.
56
+ encoder_layers (`int`, *optional*, defaults to 6):
57
+ Number of encoder layers.
58
+ encoder_ffn_dim (`int`, *optional*, defaults to 2048):
59
+ Dimension of the "intermediate" (often named feed-forward) layer in decoder.
60
+ encoder_attention_heads (`int`, *optional*, defaults to 8):
61
+ Number of attention heads for each attention layer in the Transformer encoder.
62
+ decoder_layers (`int`, *optional*, defaults to 6):
63
+ Number of decoder layers.
64
+ decoder_ffn_dim (`int`, *optional*, defaults to 2048):
65
+ Dimension of the "intermediate" (often named feed-forward) layer in decoder.
66
+ decoder_attention_heads (`int`, *optional*, defaults to 8):
67
+ Number of attention heads for each attention layer in the Transformer decoder.
68
+ is_encoder_decoder (`bool`, *optional*, defaults to `True`):
69
+ Whether the model is used as an encoder/decoder or not.
70
+ activation_function (`str` or `function`, *optional*, defaults to `"relu"`):
71
+ The non-linear activation function (function or string) in the encoder and pooler. If string, `"gelu"`,
72
+ `"relu"`, `"silu"` and `"gelu_new"` are supported.
73
+ d_model (`int`, *optional*, defaults to 256):
74
+ Dimension of the layers.
75
+ dropout (`float`, *optional*, defaults to 0.1):
76
+ The dropout probability for all fully connected layers in the embeddings, encoder, and pooler.
77
+ attention_dropout (`float`, *optional*, defaults to 0.0):
78
+ The dropout ratio for the attention probabilities.
79
+ activation_dropout (`float`, *optional*, defaults to 0.0):
80
+ The dropout ratio for activations inside the fully connected layer.
81
+ auxiliary_loss (`bool`, *optional*, defaults to `False`):
82
+ Whether auxiliary decoding losses (loss at each decoder layer) are to be used.
83
+ position_embedding_type (`str`, *optional*, defaults to `"sine"`):
84
+ Type of position embeddings to be used on top of the image features. One of `"sine"` or `"learned"`.
85
+ num_feature_levels (`int`, *optional*, defaults to 4):
86
+ The number of input feature levels.
87
+ encoder_n_points (`int`, *optional*, defaults to 4):
88
+ The number of sampled keys in each feature level for each attention head in the encoder.
89
+ decoder_n_points (`int`, *optional*, defaults to 4):
90
+ The number of sampled keys in each feature level for each attention head in the decoder.
91
+ two_stage (`bool`, *optional*, defaults to `True`):
92
+ Whether to apply a two-stage deformable DETR, where the region proposals are also generated by a variant of
93
+ Grounding DINO, which are further fed into the decoder for iterative bounding box refinement.
94
+ class_cost (`float`, *optional*, defaults to 1.0):
95
+ Relative weight of the classification error in the Hungarian matching cost.
96
+ bbox_cost (`float`, *optional*, defaults to 5.0):
97
+ Relative weight of the L1 error of the bounding box coordinates in the Hungarian matching cost.
98
+ giou_cost (`float`, *optional*, defaults to 2.0):
99
+ Relative weight of the generalized IoU loss of the bounding box in the Hungarian matching cost.
100
+ bbox_loss_coefficient (`float`, *optional*, defaults to 5.0):
101
+ Relative weight of the L1 bounding box loss in the object detection loss.
102
+ giou_loss_coefficient (`float`, *optional*, defaults to 2.0):
103
+ Relative weight of the generalized IoU loss in the object detection loss.
104
+ focal_alpha (`float`, *optional*, defaults to 0.25):
105
+ Alpha parameter in the focal loss.
106
+ disable_custom_kernels (`bool`, *optional*, defaults to `False`):
107
+ Disable the use of custom CUDA and CPU kernels. This option is necessary for the ONNX export, as custom
108
+ kernels are not supported by PyTorch ONNX export.
109
+ max_text_len (`int`, *optional*, defaults to 256):
110
+ The maximum length of the text input.
111
+ text_enhancer_dropout (`float`, *optional*, defaults to 0.0):
112
+ The dropout ratio for the text enhancer.
113
+ fusion_droppath (`float`, *optional*, defaults to 0.1):
114
+ The droppath ratio for the fusion module.
115
+ fusion_dropout (`float`, *optional*, defaults to 0.0):
116
+ The dropout ratio for the fusion module.
117
+ embedding_init_target (`bool`, *optional*, defaults to `True`):
118
+ Whether to initialize the target with Embedding weights.
119
+ query_dim (`int`, *optional*, defaults to 4):
120
+ The dimension of the query vector.
121
+ decoder_bbox_embed_share (`bool`, *optional*, defaults to `True`):
122
+ Whether to share the bbox regression head for all decoder layers.
123
+ two_stage_bbox_embed_share (`bool`, *optional*, defaults to `False`):
124
+ Whether to share the bbox embedding between the two-stage bbox generator and the region proposal
125
+ generation.
126
+ positional_embedding_temperature (`float`, *optional*, defaults to 20):
127
+ The temperature for Sine Positional Embedding that is used together with vision backbone.
128
+ init_std (`float`, *optional*, defaults to 0.02):
129
+ The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
130
+ layer_norm_eps (`float`, *optional*, defaults to 1e-05):
131
+ The epsilon used by the layer normalization layers.
132
+
133
+ Examples:
134
+
135
+ ```python
136
+ >>> from transformers import GroundingDinoConfig, GroundingDinoModel
137
+
138
+ >>> # Initializing a Grounding DINO IDEA-Research/grounding-dino-tiny style configuration
139
+ >>> configuration = GroundingDinoConfig()
140
+
141
+ >>> # Initializing a model (with random weights) from the IDEA-Research/grounding-dino-tiny style configuration
142
+ >>> model = GroundingDinoModel(configuration)
143
+
144
+ >>> # Accessing the model configuration
145
+ >>> configuration = model.config
146
+ ```"""
147
+
148
+ model_type = "grounding-dino"
149
+ attribute_map = {
150
+ "hidden_size": "d_model",
151
+ "num_attention_heads": "encoder_attention_heads",
152
+ }
153
+
154
+ def __init__(
155
+ self,
156
+ backbone_config=None,
157
+ backbone=None,
158
+ use_pretrained_backbone=False,
159
+ use_timm_backbone=False,
160
+ backbone_kwargs=None,
161
+ text_config=None,
162
+ num_queries=900,
163
+ encoder_layers=6,
164
+ encoder_ffn_dim=2048,
165
+ encoder_attention_heads=8,
166
+ decoder_layers=6,
167
+ decoder_ffn_dim=2048,
168
+ decoder_attention_heads=8,
169
+ is_encoder_decoder=True,
170
+ activation_function="relu",
171
+ d_model=256,
172
+ dropout=0.1,
173
+ attention_dropout=0.0,
174
+ activation_dropout=0.0,
175
+ auxiliary_loss=False,
176
+ position_embedding_type="sine",
177
+ num_feature_levels=4,
178
+ encoder_n_points=4,
179
+ decoder_n_points=4,
180
+ two_stage=True,
181
+ class_cost=1.0,
182
+ bbox_cost=5.0,
183
+ giou_cost=2.0,
184
+ bbox_loss_coefficient=5.0,
185
+ giou_loss_coefficient=2.0,
186
+ focal_alpha=0.25,
187
+ disable_custom_kernels=False,
188
+ # other parameters
189
+ max_text_len=256,
190
+ text_enhancer_dropout=0.0,
191
+ fusion_droppath=0.1,
192
+ fusion_dropout=0.0,
193
+ embedding_init_target=True,
194
+ query_dim=4,
195
+ decoder_bbox_embed_share=True,
196
+ two_stage_bbox_embed_share=False,
197
+ positional_embedding_temperature=20,
198
+ init_std=0.02,
199
+ layer_norm_eps=1e-5,
200
+ **kwargs,
201
+ ):
202
+ if backbone_config is None and backbone is None:
203
+ logger.info("`backbone_config` is `None`. Initializing the config with the default `Swin` backbone.")
204
+ backbone_config = CONFIG_MAPPING["swin"](
205
+ window_size=7,
206
+ image_size=224,
207
+ embed_dim=96,
208
+ depths=[2, 2, 6, 2],
209
+ num_heads=[3, 6, 12, 24],
210
+ out_indices=[2, 3, 4],
211
+ )
212
+ elif isinstance(backbone_config, dict):
213
+ backbone_model_type = backbone_config.pop("model_type")
214
+ config_class = CONFIG_MAPPING[backbone_model_type]
215
+ backbone_config = config_class.from_dict(backbone_config)
216
+
217
+ verify_backbone_config_arguments(
218
+ use_timm_backbone=use_timm_backbone,
219
+ use_pretrained_backbone=use_pretrained_backbone,
220
+ backbone=backbone,
221
+ backbone_config=backbone_config,
222
+ backbone_kwargs=backbone_kwargs,
223
+ )
224
+
225
+ if text_config is None:
226
+ text_config = {}
227
+ logger.info("text_config is None. Initializing the text config with default values (`BertConfig`).")
228
+
229
+ self.backbone_config = backbone_config
230
+ self.backbone = backbone
231
+ self.use_pretrained_backbone = use_pretrained_backbone
232
+ self.use_timm_backbone = use_timm_backbone
233
+ self.backbone_kwargs = backbone_kwargs
234
+ self.num_queries = num_queries
235
+ self.d_model = d_model
236
+ self.encoder_ffn_dim = encoder_ffn_dim
237
+ self.encoder_layers = encoder_layers
238
+ self.encoder_attention_heads = encoder_attention_heads
239
+ self.decoder_ffn_dim = decoder_ffn_dim
240
+ self.decoder_layers = decoder_layers
241
+ self.decoder_attention_heads = decoder_attention_heads
242
+ self.dropout = dropout
243
+ self.attention_dropout = attention_dropout
244
+ self.activation_dropout = activation_dropout
245
+ self.activation_function = activation_function
246
+ self.auxiliary_loss = auxiliary_loss
247
+ self.position_embedding_type = position_embedding_type
248
+ # deformable attributes
249
+ self.num_feature_levels = num_feature_levels
250
+ self.encoder_n_points = encoder_n_points
251
+ self.decoder_n_points = decoder_n_points
252
+ self.two_stage = two_stage
253
+ # Hungarian matcher
254
+ self.class_cost = class_cost
255
+ self.bbox_cost = bbox_cost
256
+ self.giou_cost = giou_cost
257
+ # Loss coefficients
258
+ self.bbox_loss_coefficient = bbox_loss_coefficient
259
+ self.giou_loss_coefficient = giou_loss_coefficient
260
+ self.focal_alpha = focal_alpha
261
+ self.disable_custom_kernels = disable_custom_kernels
262
+ # Text backbone
263
+ if isinstance(text_config, dict):
264
+ text_config["model_type"] = text_config.get("model_type", "bert")
265
+ text_config = CONFIG_MAPPING[text_config["model_type"]](**text_config)
266
+ elif text_config is None:
267
+ text_config = CONFIG_MAPPING["bert"]()
268
+
269
+ self.text_config = text_config
270
+ self.max_text_len = max_text_len
271
+
272
+ # Text Enhancer
273
+ self.text_enhancer_dropout = text_enhancer_dropout
274
+ # Fusion
275
+ self.fusion_droppath = fusion_droppath
276
+ self.fusion_dropout = fusion_dropout
277
+ # Others
278
+ self.embedding_init_target = embedding_init_target
279
+ self.query_dim = query_dim
280
+ self.decoder_bbox_embed_share = decoder_bbox_embed_share
281
+ self.two_stage_bbox_embed_share = two_stage_bbox_embed_share
282
+ if two_stage_bbox_embed_share and not decoder_bbox_embed_share:
283
+ raise ValueError("If two_stage_bbox_embed_share is True, decoder_bbox_embed_share must be True.")
284
+ self.positional_embedding_temperature = positional_embedding_temperature
285
+ self.init_std = init_std
286
+ self.layer_norm_eps = layer_norm_eps
287
+ super().__init__(is_encoder_decoder=is_encoder_decoder, **kwargs)
288
+
289
+ @property
290
+ def num_attention_heads(self) -> int:
291
+ return self.encoder_attention_heads
292
+
293
+ @property
294
+ def hidden_size(self) -> int:
295
+ return self.d_model
296
+
297
+ @property
298
+ def sub_configs(self):
299
+ sub_configs = {}
300
+ backbone_config = getattr(self, "backbone_config", None)
301
+ text_config = getattr(self, "text_config", None)
302
+ if isinstance(backbone_config, PretrainedConfig):
303
+ sub_configs["backbone_config"] = type(backbone_config)
304
+ if isinstance(text_config, PretrainedConfig):
305
+ sub_configs["text_config"] = type(self.text_config)
306
+ return sub_configs
307
+
308
+
309
+ __all__ = ["GroundingDinoConfig"]