Buckets:

HuggingFaceDocBuilder's picture
download
raw
71.6 kB
import{s as Ot,b as Pt,o as Kt,n as Ze}from"../chunks/scheduler.31fdf58d.js";import{S as eo,i as to,e as i,s,c as u,h as oo,a as d,d as o,b as a,f as ge,j as h,g as f,k as R,l as c,m as n,n as g,t as b,o as _,p as y}from"../chunks/index.2f76fdf0.js";import{T as Lt}from"../chunks/Tip.8d349121.js";import{C as no}from"../chunks/CopyLLMTxtMenu.6ffa41dc.js";import{D as $e}from"../chunks/Docstring.ac2bce02.js";import{C as V}from"../chunks/CodeBlock.e52df5d6.js";import{E as Ut}from"../chunks/ExampleCodeBlock.f8b50594.js";import{H as xe,E as so}from"../chunks/MermaidChart.svelte_svelte_type_style_lang.66a55e6f.js";function ao(Z){let r,w="Examples:",p,m,M;return m=new V({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMERhYkRldHJDb25maWclMkMlMjBEYWJEZXRyTW9kZWwlMEElMEElMjMlMjBJbml0aWFsaXppbmclMjBhJTIwREFCLURFVFIlMjBJREVBLVJlc2VhcmNoJTJGZGFiLWRldHItcmVzbmV0LTUwJTIwc3R5bGUlMjBjb25maWd1cmF0aW9uJTBBY29uZmlndXJhdGlvbiUyMCUzRCUyMERhYkRldHJDb25maWcoKSUwQSUwQSUyMyUyMEluaXRpYWxpemluZyUyMGElMjBtb2RlbCUyMCh3aXRoJTIwcmFuZG9tJTIwd2VpZ2h0cyklMjBmcm9tJTIwdGhlJTIwSURFQS1SZXNlYXJjaCUyRmRhYi1kZXRyLXJlc25ldC01MCUyMHN0eWxlJTIwY29uZmlndXJhdGlvbiUwQW1vZGVsJTIwJTNEJTIwRGFiRGV0ck1vZGVsKGNvbmZpZ3VyYXRpb24pJTBBJTBBJTIzJTIwQWNjZXNzaW5nJTIwdGhlJTIwbW9kZWwlMjBjb25maWd1cmF0aW9uJTBBY29uZmlndXJhdGlvbiUyMCUzRCUyMG1vZGVsLmNvbmZpZw==",highlighted:`<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> DabDetrConfig, DabDetrModel
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-comment"># Initializing a DAB-DETR IDEA-Research/dab-detr-resnet-50 style configuration</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>configuration = DabDetrConfig()
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-comment"># Initializing a model (with random weights) from the IDEA-Research/dab-detr-resnet-50 style configuration</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>model = DabDetrModel(configuration)
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-comment"># Accessing the model configuration</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>configuration = model.config`,lang:"python",wrap:!1}}),{c(){r=i("p"),r.textContent=w,p=s(),u(m.$$.fragment)},l(l){r=d(l,"P",{"data-svelte-h":!0}),h(r)!=="svelte-kvfsh7"&&(r.textContent=w),p=a(l),f(m.$$.fragment,l)},m(l,T){n(l,r,T),n(l,p,T),g(m,l,T),M=!0},p:Ze,i(l){M||(b(m.$$.fragment,l),M=!0)},o(l){_(m.$$.fragment,l),M=!1},d(l){l&&(o(r),o(p)),y(m,l)}}}function ro(Z){let r,w=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code>
instance afterwards instead of this since the former takes care of running the pre and post processing steps while
the latter silently ignores them.`;return{c(){r=i("p"),r.innerHTML=w},l(p){r=d(p,"P",{"data-svelte-h":!0}),h(r)!=="svelte-fincs2"&&(r.innerHTML=w)},m(p,m){n(p,r,m)},p:Ze,d(p){p&&o(r)}}}function lo(Z){let r,w="Examples:",p,m,M;return m=new V({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9JbWFnZVByb2Nlc3NvciUyQyUyMEF1dG9Nb2RlbCUwQWZyb20lMjBQSUwlMjBpbXBvcnQlMjBJbWFnZSUwQWltcG9ydCUyMGh0dHB4JTBBZnJvbSUyMGlvJTIwaW1wb3J0JTIwQnl0ZXNJTyUwQSUwQXVybCUyMCUzRCUyMCUyMmh0dHAlM0ElMkYlMkZpbWFnZXMuY29jb2RhdGFzZXQub3JnJTJGdmFsMjAxNyUyRjAwMDAwMDAzOTc2OS5qcGclMjIlMEF3aXRoJTIwaHR0cHguc3RyZWFtKCUyMkdFVCUyMiUyQyUyMHVybCklMjBhcyUyMHJlc3BvbnNlJTNBJTBBJTIwJTIwJTIwJTIwaW1hZ2UlMjAlM0QlMjBJbWFnZS5vcGVuKEJ5dGVzSU8ocmVzcG9uc2UucmVhZCgpKSklMEElMEFpbWFnZV9wcm9jZXNzb3IlMjAlM0QlMjBBdXRvSW1hZ2VQcm9jZXNzb3IuZnJvbV9wcmV0cmFpbmVkKCUyMklERUEtUmVzZWFyY2glMkZkYWItZGV0ci1yZXNuZXQtNTAlMjIpJTBBbW9kZWwlMjAlM0QlMjBBdXRvTW9kZWwuZnJvbV9wcmV0cmFpbmVkKCUyMklERUEtUmVzZWFyY2glMkZkYWItZGV0ci1yZXNuZXQtNTAlMjIpJTBBJTBBJTIzJTIwcHJlcGFyZSUyMGltYWdlJTIwZm9yJTIwdGhlJTIwbW9kZWwlMEFpbnB1dHMlMjAlM0QlMjBpbWFnZV9wcm9jZXNzb3IoaW1hZ2VzJTNEaW1hZ2UlMkMlMjByZXR1cm5fdGVuc29ycyUzRCUyMnB0JTIyKSUwQSUwQSUyMyUyMGZvcndhcmQlMjBwYXNzJTBBb3V0cHV0cyUyMCUzRCUyMG1vZGVsKCoqaW5wdXRzKSUwQSUwQSUyMyUyMHRoZSUyMGxhc3QlMjBoaWRkZW4lMjBzdGF0ZXMlMjBhcmUlMjB0aGUlMjBmaW5hbCUyMHF1ZXJ5JTIwZW1iZWRkaW5ncyUyMG9mJTIwdGhlJTIwVHJhbnNmb3JtZXIlMjBkZWNvZGVyJTBBJTIzJTIwdGhlc2UlMjBhcmUlMjBvZiUyMHNoYXBlJTIwKGJhdGNoX3NpemUlMkMlMjBudW1fcXVlcmllcyUyQyUyMGhpZGRlbl9zaXplKSUwQWxhc3RfaGlkZGVuX3N0YXRlcyUyMCUzRCUyMG91dHB1dHMubGFzdF9oaWRkZW5fc3RhdGUlMEFsaXN0KGxhc3RfaGlkZGVuX3N0YXRlcy5zaGFwZSk=",highlighted:`<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoImageProcessor, AutoModel
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> PIL <span class="hljs-keyword">import</span> Image
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">import</span> httpx
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> io <span class="hljs-keyword">import</span> BytesIO
<span class="hljs-meta">&gt;&gt;&gt; </span>url = <span class="hljs-string">&quot;http://images.cocodataset.org/val2017/000000039769.jpg&quot;</span>
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">with</span> httpx.stream(<span class="hljs-string">&quot;GET&quot;</span>, url) <span class="hljs-keyword">as</span> response:
<span class="hljs-meta">... </span> image = Image.<span class="hljs-built_in">open</span>(BytesIO(response.read()))
<span class="hljs-meta">&gt;&gt;&gt; </span>image_processor = AutoImageProcessor.from_pretrained(<span class="hljs-string">&quot;IDEA-Research/dab-detr-resnet-50&quot;</span>)
<span class="hljs-meta">&gt;&gt;&gt; </span>model = AutoModel.from_pretrained(<span class="hljs-string">&quot;IDEA-Research/dab-detr-resnet-50&quot;</span>)
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-comment"># prepare image for the model</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>inputs = image_processor(images=image, return_tensors=<span class="hljs-string">&quot;pt&quot;</span>)
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-comment"># forward pass</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>outputs = model(**inputs)
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-comment"># the last hidden states are the final query embeddings of the Transformer decoder</span>
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-comment"># these are of shape (batch_size, num_queries, hidden_size)</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>last_hidden_states = outputs.last_hidden_state
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-built_in">list</span>(last_hidden_states.shape)
[<span class="hljs-number">1</span>, <span class="hljs-number">300</span>, <span class="hljs-number">256</span>]`,lang:"python",wrap:!1}}),{c(){r=i("p"),r.textContent=w,p=s(),u(m.$$.fragment)},l(l){r=d(l,"P",{"data-svelte-h":!0}),h(r)!=="svelte-kvfsh7"&&(r.textContent=w),p=a(l),f(m.$$.fragment,l)},m(l,T){n(l,r,T),n(l,p,T),g(m,l,T),M=!0},p:Ze,i(l){M||(b(m.$$.fragment,l),M=!0)},o(l){_(m.$$.fragment,l),M=!1},d(l){l&&(o(r),o(p)),y(m,l)}}}function io(Z){let r,w=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code>
instance afterwards instead of this since the former takes care of running the pre and post processing steps while
the latter silently ignores them.`;return{c(){r=i("p"),r.innerHTML=w},l(p){r=d(p,"P",{"data-svelte-h":!0}),h(r)!=="svelte-fincs2"&&(r.innerHTML=w)},m(p,m){n(p,r,m)},p:Ze,d(p){p&&o(r)}}}function co(Z){let r,w="Examples:",p,m,M;return m=new V({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9JbWFnZVByb2Nlc3NvciUyQyUyMEF1dG9Nb2RlbEZvck9iamVjdERldGVjdGlvbiUwQWZyb20lMjBQSUwlMjBpbXBvcnQlMjBJbWFnZSUwQWltcG9ydCUyMGh0dHB4JTBBZnJvbSUyMGlvJTIwaW1wb3J0JTIwQnl0ZXNJTyUwQSUwQXVybCUyMCUzRCUyMCUyMmh0dHAlM0ElMkYlMkZpbWFnZXMuY29jb2RhdGFzZXQub3JnJTJGdmFsMjAxNyUyRjAwMDAwMDAzOTc2OS5qcGclMjIlMEF3aXRoJTIwaHR0cHguc3RyZWFtKCUyMkdFVCUyMiUyQyUyMHVybCklMjBhcyUyMHJlc3BvbnNlJTNBJTBBJTIwJTIwJTIwJTIwaW1hZ2UlMjAlM0QlMjBJbWFnZS5vcGVuKEJ5dGVzSU8ocmVzcG9uc2UucmVhZCgpKSklMEElMEFpbWFnZV9wcm9jZXNzb3IlMjAlM0QlMjBBdXRvSW1hZ2VQcm9jZXNzb3IuZnJvbV9wcmV0cmFpbmVkKCUyMklERUEtUmVzZWFyY2glMkZkYWItZGV0ci1yZXNuZXQtNTAlMjIpJTBBbW9kZWwlMjAlM0QlMjBBdXRvTW9kZWxGb3JPYmplY3REZXRlY3Rpb24uZnJvbV9wcmV0cmFpbmVkKCUyMklERUEtUmVzZWFyY2glMkZkYWItZGV0ci1yZXNuZXQtNTAlMjIpJTBBJTBBaW5wdXRzJTIwJTNEJTIwaW1hZ2VfcHJvY2Vzc29yKGltYWdlcyUzRGltYWdlJTJDJTIwcmV0dXJuX3RlbnNvcnMlM0QlMjJwdCUyMiklMEElMEF3aXRoJTIwdG9yY2gubm9fZ3JhZCgpJTNBJTBBJTIwJTIwJTIwJTIwb3V0cHV0cyUyMCUzRCUyMG1vZGVsKCoqaW5wdXRzKSUwQSUwQSUyMyUyMGNvbnZlcnQlMjBvdXRwdXRzJTIwKGJvdW5kaW5nJTIwYm94ZXMlMjBhbmQlMjBjbGFzcyUyMGxvZ2l0cyklMjB0byUyMFBhc2NhbCUyMFZPQyUyMGZvcm1hdCUyMCh4bWluJTJDJTIweW1pbiUyQyUyMHhtYXglMkMlMjB5bWF4KSUwQXRhcmdldF9zaXplcyUyMCUzRCUyMHRvcmNoLnRlbnNvciglNUIoaW1hZ2UuaGVpZ2h0JTJDJTIwaW1hZ2Uud2lkdGgpJTVEKSUwQXJlc3VsdHMlMjAlM0QlMjBpbWFnZV9wcm9jZXNzb3IucG9zdF9wcm9jZXNzX29iamVjdF9kZXRlY3Rpb24ob3V0cHV0cyUyQyUyMHRocmVzaG9sZCUzRDAuNSUyQyUyMHRhcmdldF9zaXplcyUzRHRhcmdldF9zaXplcyklNUIwJTVEJTBBZm9yJTIwc2NvcmUlMkMlMjBsYWJlbCUyQyUyMGJveCUyMGluJTIwemlwKHJlc3VsdHMlNUIlMjJzY29yZXMlMjIlNUQlMkMlMjByZXN1bHRzJTVCJTIybGFiZWxzJTIyJTVEJTJDJTIwcmVzdWx0cyU1QiUyMmJveGVzJTIyJTVEKSUzQSUwQSUyMCUyMCUyMCUyMGJveCUyMCUzRCUyMCU1QnJvdW5kKGklMkMlMjAyKSUyMGZvciUyMGklMjBpbiUyMGJveC50b2xpc3QoKSU1RCUwQSUyMCUyMCUyMCUyMHByaW50KCUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMGYlMjJEZXRlY3RlZCUyMCU3Qm1vZGVsLmNvbmZpZy5pZDJsYWJlbCU1QmxhYmVsLml0ZW0oKSU1RCU3RCUyMHdpdGglMjBjb25maWRlbmNlJTIwJTIyJTBBJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwZiUyMiU3QnJvdW5kKHNjb3JlLml0ZW0oKSUyQyUyMDMpJTdEJTIwYXQlMjBsb2NhdGlvbiUyMCU3QmJveCU3RCUyMiUwQSUyMCUyMCUyMCUyMCk=",highlighted:`<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoImageProcessor, AutoModelForObjectDetection
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> PIL <span class="hljs-keyword">import</span> Image
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">import</span> httpx
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">from</span> io <span class="hljs-keyword">import</span> BytesIO
<span class="hljs-meta">&gt;&gt;&gt; </span>url = <span class="hljs-string">&quot;http://images.cocodataset.org/val2017/000000039769.jpg&quot;</span>
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">with</span> httpx.stream(<span class="hljs-string">&quot;GET&quot;</span>, url) <span class="hljs-keyword">as</span> response:
<span class="hljs-meta">... </span> image = Image.<span class="hljs-built_in">open</span>(BytesIO(response.read()))
<span class="hljs-meta">&gt;&gt;&gt; </span>image_processor = AutoImageProcessor.from_pretrained(<span class="hljs-string">&quot;IDEA-Research/dab-detr-resnet-50&quot;</span>)
<span class="hljs-meta">&gt;&gt;&gt; </span>model = AutoModelForObjectDetection.from_pretrained(<span class="hljs-string">&quot;IDEA-Research/dab-detr-resnet-50&quot;</span>)
<span class="hljs-meta">&gt;&gt;&gt; </span>inputs = image_processor(images=image, return_tensors=<span class="hljs-string">&quot;pt&quot;</span>)
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">with</span> torch.no_grad():
<span class="hljs-meta">&gt;&gt;&gt; </span> outputs = model(**inputs)
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-comment"># convert outputs (bounding boxes and class logits) to Pascal VOC format (xmin, ymin, xmax, ymax)</span>
<span class="hljs-meta">&gt;&gt;&gt; </span>target_sizes = torch.tensor([(image.height, image.width)])
<span class="hljs-meta">&gt;&gt;&gt; </span>results = image_processor.post_process_object_detection(outputs, threshold=<span class="hljs-number">0.5</span>, target_sizes=target_sizes)[<span class="hljs-number">0</span>]
<span class="hljs-meta">&gt;&gt;&gt; </span><span class="hljs-keyword">for</span> score, label, box <span class="hljs-keyword">in</span> <span class="hljs-built_in">zip</span>(results[<span class="hljs-string">&quot;scores&quot;</span>], results[<span class="hljs-string">&quot;labels&quot;</span>], results[<span class="hljs-string">&quot;boxes&quot;</span>]):
<span class="hljs-meta">... </span> box = [<span class="hljs-built_in">round</span>(i, <span class="hljs-number">2</span>) <span class="hljs-keyword">for</span> i <span class="hljs-keyword">in</span> box.tolist()]
<span class="hljs-meta">... </span> <span class="hljs-built_in">print</span>(
<span class="hljs-meta">... </span> <span class="hljs-string">f&quot;Detected <span class="hljs-subst">{model.config.id2label[label.item()]}</span> with confidence &quot;</span>
<span class="hljs-meta">... </span> <span class="hljs-string">f&quot;<span class="hljs-subst">{<span class="hljs-built_in">round</span>(score.item(), <span class="hljs-number">3</span>)}</span> at location <span class="hljs-subst">{box}</span>&quot;</span>
<span class="hljs-meta">... </span> )
Detected remote <span class="hljs-keyword">with</span> confidence <span class="hljs-number">0.833</span> at location [<span class="hljs-number">38.31</span>, <span class="hljs-number">72.1</span>, <span class="hljs-number">177.63</span>, <span class="hljs-number">118.45</span>]
Detected cat <span class="hljs-keyword">with</span> confidence <span class="hljs-number">0.831</span> at location [<span class="hljs-number">9.2</span>, <span class="hljs-number">51.38</span>, <span class="hljs-number">321.13</span>, <span class="hljs-number">469.0</span>]
Detected cat <span class="hljs-keyword">with</span> confidence <span class="hljs-number">0.804</span> at location [<span class="hljs-number">340.3</span>, <span class="hljs-number">16.85</span>, <span class="hljs-number">642.93</span>, <span class="hljs-number">370.95</span>]
Detected remote <span class="hljs-keyword">with</span> confidence <span class="hljs-number">0.683</span> at location [<span class="hljs-number">334.48</span>, <span class="hljs-number">73.49</span>, <span class="hljs-number">366.37</span>, <span class="hljs-number">190.01</span>]
Detected couch <span class="hljs-keyword">with</span> confidence <span class="hljs-number">0.535</span> at location [<span class="hljs-number">0.52</span>, <span class="hljs-number">1.19</span>, <span class="hljs-number">640.35</span>, <span class="hljs-number">475.1</span>]`,lang:"python",wrap:!1}}),{c(){r=i("p"),r.textContent=w,p=s(),u(m.$$.fragment)},l(l){r=d(l,"P",{"data-svelte-h":!0}),h(r)!=="svelte-kvfsh7"&&(r.textContent=w),p=a(l),f(m.$$.fragment,l)},m(l,T){n(l,r,T),n(l,p,T),g(m,l,T),M=!0},p:Ze,i(l){M||(b(m.$$.fragment,l),M=!0)},o(l){_(m.$$.fragment,l),M=!1},d(l){l&&(o(r),o(p)),y(m,l)}}}function po(Z){let r,w,p,m,M,l="<em>This model was published in HF papers on 2022-01-28 and contributed to Hugging Face Transformers on 2025-02-04.</em>",T,G,ze,q,Re,Q,Ie,X,Ct=`The DAB-DETR model was proposed in <a href="https://huggingface.co/papers/2201.12329" rel="nofollow">DAB-DETR: Dynamic Anchor Boxes are Better Queries for DETR</a> by Shilong Liu, Feng Li, Hao Zhang, Xiao Yang, Xianbiao Qi, Hang Su, Jun Zhu, Lei Zhang.
DAB-DETR is an enhanced variant of Conditional DETR. It utilizes dynamically updated anchor boxes to provide both a reference query point (x, y) and a reference anchor size (w, h), improving cross-attention computation. This new approach achieves 45.7% AP when trained for 50 epochs with a single ResNet-50 model as the backbone.`,Fe,I,Jt,We,H,xt="The abstract from the paper is the following:",Ne,A,kt=`<em>We present in this paper a novel query formulation using dynamic anchor boxes
for DETR (DEtection TRansformer) and offer a deeper understanding of the role
of queries in DETR. This new formulation directly uses box coordinates as queries
in Transformer decoders and dynamically updates them layer-by-layer. Using box
coordinates not only helps using explicit positional priors to improve the query-to-feature similarity and eliminate the slow training convergence issue in DETR,
but also allows us to modulate the positional attention map using the box width
and height information. Such a design makes it clear that queries in DETR can be
implemented as performing soft ROI pooling layer-by-layer in a cascade manner.
As a result, it leads to the best performance on MS-COCO benchmark among
the DETR-like detection models under the same setting, e.g., AP 45.7% using
ResNet50-DC5 as backbone trained in 50 epochs. We also conducted extensive
experiments to confirm our analysis and verify the effectiveness of our methods.</em>`,Be,S,$t=`This model was contributed by <a href="https://huggingface.co/davidhajdu" rel="nofollow">davidhajdu</a>.
The original code can be found <a href="https://github.com/IDEA-Research/DAB-DETR" rel="nofollow">here</a>.`,Ee,Y,Ve,L,Zt="Use the code below to get started with the model.",Ge,O,qe,P,zt="This should output",Qe,K,Xe,ee,Rt="There are three other ways to instantiate a DAB-DETR model (depending on what you prefer):",He,te,It="Option 1: Instantiate DAB-DETR with pre-trained weights for entire model",Ae,oe,Se,ne,Ft="Option 2: Instantiate DAB-DETR with randomly initialized weights for Transformer, but pre-trained weights for backbone",Ye,se,Le,ae,Wt="Option 3: Instantiate DAB-DETR with randomly initialized weights for backbone + Transformer",Oe,re,Pe,le,Ke,C,ie,lt,be,Nt=`This is the configuration class to store the configuration of a Dab DetrModel. It is used to instantiate a Dab Detr
model according to the specified arguments, defining the model architecture. Instantiating a configuration with the
defaults will yield a similar configuration to that of the <a href="https://huggingface.co/IDEA-Research/dab-detr-resnet-50" rel="nofollow">IDEA-Research/dab-detr-resnet-50</a>`,it,_e,Bt=`Configuration objects inherit from <a href="/docs/transformers/pr_41116/en/main_classes/configuration#transformers.PreTrainedConfig">PreTrainedConfig</a> and can be used to control the model outputs. Read the
documentation from <a href="/docs/transformers/pr_41116/en/main_classes/configuration#transformers.PreTrainedConfig">PreTrainedConfig</a> for more information.`,dt,F,et,de,tt,j,ce,ct,ye,Et=`The bare DAB-DETR Model (consisting of a backbone and encoder-decoder Transformer) outputting raw
hidden-states, intermediate hidden states, reference points, output coordinates without any specific head on top.`,pt,Me,Vt=`This model inherits from <a href="/docs/transformers/pr_41116/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the
library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads
etc.)`,mt,we,Gt=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass.
Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage
and behavior.`,ht,v,pe,ut,Te,qt='The <a href="/docs/transformers/pr_41116/en/model_doc/dab-detr#transformers.DabDetrModel">DabDetrModel</a> forward method, overrides the <code>__call__</code> special method.',ft,W,gt,je,Qt=`<li><p><strong>last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>) — Sequence of hidden-states at the output of the last layer of the decoder of the model.</p> <p>If <code>past_key_values</code> is used only the last hidden-state of the sequences of shape <code>(batch_size, 1, hidden_size)</code> is output.</p></li> <li><p><strong>past_key_values</strong> (<code>EncoderDecoderCache</code>, <em>optional</em>, returned when <code>use_cache=True</code> is passed or when <code>config.use_cache=True</code>) — It is a <a href="/docs/transformers/pr_41116/en/internal/generation_utils#transformers.EncoderDecoderCache">EncoderDecoderCache</a> instance. For more details, see our <a href="https://huggingface.co/docs/transformers/en/kv_cache" rel="nofollow">kv cache guide</a>.</p> <p>Contains pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention
blocks) that can be used (see <code>past_key_values</code> input) to speed up sequential decoding.</p></li> <li><p><strong>decoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, +
one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> <p>Hidden-states of the decoder at the output of each layer plus the optional initial embedding outputs.</p></li> <li><p><strong>decoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> <p>Attentions weights of the decoder, after the attention softmax, used to compute the weighted average in the
self-attention heads.</p></li> <li><p><strong>cross_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> <p>Attentions weights of the decoder’s cross-attention layer, after the attention softmax, used to compute the
weighted average in the cross-attention heads.</p></li> <li><p><strong>encoder_last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — Sequence of hidden-states at the output of the last layer of the encoder of the model.</p></li> <li><p><strong>encoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, +
one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> <p>Hidden-states of the encoder at the output of each layer plus the optional initial embedding outputs.</p></li> <li><p><strong>encoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> <p>Attentions weights of the encoder, after the attention softmax, used to compute the weighted average in the
self-attention heads.</p></li> <li><p><strong>intermediate_hidden_states</strong> (<code>torch.FloatTensor</code> of shape <code>(config.decoder_layers, batch_size, sequence_length, hidden_size)</code>, <em>optional</em>, returned when <code>config.auxiliary_loss=True</code>) — Intermediate decoder activations, i.e. the output of each decoder layer, each of them gone through a
layernorm.</p></li> <li><p><strong>reference_points</strong> (<code>torch.FloatTensor</code> of shape <code>(config.decoder_layers, batch_size, num_queries, 2 (anchor points))</code>) — Reference points (reference points of each layer of the decoder).</p></li>`,bt,N,ot,me,nt,D,he,_t,De,Xt=`DAB_DETR Model (consisting of a backbone and encoder-decoder Transformer) with object detection heads on
top, for tasks such as COCO detection.`,yt,ve,Ht=`This model inherits from <a href="/docs/transformers/pr_41116/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the
library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads
etc.)`,Mt,Ue,At=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass.
Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage
and behavior.`,wt,U,ue,Tt,Ce,St='The <a href="/docs/transformers/pr_41116/en/model_doc/dab-detr#transformers.DabDetrForObjectDetection">DabDetrForObjectDetection</a> forward method, overrides the <code>__call__</code> special method.',jt,B,Dt,Je,Yt=`<li><p><strong>loss</strong> (<code>torch.FloatTensor</code> of shape <code>(1,)</code>, <em>optional</em>, returned when <code>labels</code> are provided)) — Total loss as a linear combination of a negative log-likehood (cross-entropy) for class prediction and a
bounding box loss. The latter is defined as a linear combination of the L1 loss and the generalized
scale-invariant IoU loss.</p></li> <li><p><strong>loss_dict</strong> (<code>Dict</code>, <em>optional</em>) — A dictionary containing the individual losses. Useful for logging.</p></li> <li><p><strong>logits</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, num_classes + 1)</code>) — Classification logits (including no-object) for all queries.</p></li> <li><p><strong>pred_boxes</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, 4)</code>) — Normalized boxes coordinates for all queries, represented as (center_x, center_y, width, height). These
values are normalized in [0, 1], relative to the size of each individual image in the batch (disregarding
possible padding). You can use <code>~DabDetrImageProcessor.post_process_object_detection</code> to retrieve the
unnormalized bounding boxes.</p></li> <li><p><strong>auxiliary_outputs</strong> (<code>list[Dict]</code>, <em>optional</em>) — Optional, only returned when auxiliary losses are activated (i.e. <code>config.auxiliary_loss</code> is set to <code>True</code>)
and labels are provided. It is a list of dictionaries containing the two above keys (<code>logits</code> and
<code>pred_boxes</code>) for each decoder layer.</p></li> <li><p><strong>last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — Sequence of hidden-states at the output of the last layer of the decoder of the model.</p></li> <li><p><strong>decoder_hidden_states</strong> (<code>tuple[torch.FloatTensor]</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, +
one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> <p>Hidden-states of the decoder at the output of each layer plus the initial embedding outputs.</p></li> <li><p><strong>decoder_attentions</strong> (<code>tuple[torch.FloatTensor]</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> <p>Attentions weights of the decoder, after the attention softmax, used to compute the weighted average in the
self-attention heads.</p></li> <li><p><strong>cross_attentions</strong> (<code>tuple[torch.FloatTensor]</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> <p>Attentions weights of the decoder’s cross-attention layer, after the attention softmax, used to compute the
weighted average in the cross-attention heads.</p></li> <li><p><strong>encoder_last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — Sequence of hidden-states at the output of the last layer of the encoder of the model.</p></li> <li><p><strong>encoder_hidden_states</strong> (<code>tuple[torch.FloatTensor]</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings, if the model has an embedding layer, +
one for the output of each layer) of shape <code>(batch_size, sequence_length, hidden_size)</code>.</p> <p>Hidden-states of the encoder at the output of each layer plus the initial embedding outputs.</p></li> <li><p><strong>encoder_attentions</strong> (<code>tuple[torch.FloatTensor]</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>.</p> <p>Attentions weights of the encoder, after the attention softmax, used to compute the weighted average in the
self-attention heads.</p></li>`,vt,E,st,fe,at,ke,rt;return G=new no({props:{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"}}),q=new xe({props:{title:"DAB-DETR",local:"dab-detr",headingTag:"h1"}}),Q=new xe({props:{title:"Overview",local:"overview",headingTag:"h2"}}),Y=new xe({props:{title:"How to Get Started with the Model",local:"how-to-get-started-with-the-model",headingTag:"h2"}}),O=new V({props:{code:"aW1wb3J0JTIwcmVxdWVzdHMlMEFpbXBvcnQlMjB0b3JjaCUwQWZyb20lMjBQSUwlMjBpbXBvcnQlMjBJbWFnZSUwQSUwQWZyb20lMjB0cmFuc2Zvcm1lcnMlMjBpbXBvcnQlMjBBdXRvSW1hZ2VQcm9jZXNzb3IlMkMlMjBBdXRvTW9kZWxGb3JPYmplY3REZXRlY3Rpb24lMEElMEElMEF1cmwlMjAlM0QlMjAnaHR0cCUzQSUyRiUyRmltYWdlcy5jb2NvZGF0YXNldC5vcmclMkZ2YWwyMDE3JTJGMDAwMDAwMDM5NzY5LmpwZyclMEFpbWFnZSUyMCUzRCUyMEltYWdlLm9wZW4ocmVxdWVzdHMuZ2V0KHVybCUyQyUyMHN0cmVhbSUzRFRydWUpLnJhdyklMEElMEFpbWFnZV9wcm9jZXNzb3IlMjAlM0QlMjBBdXRvSW1hZ2VQcm9jZXNzb3IuZnJvbV9wcmV0cmFpbmVkKCUyMklERUEtUmVzZWFyY2glMkZkYWItZGV0ci1yZXNuZXQtNTAlMjIpJTBBbW9kZWwlMjAlM0QlMjBBdXRvTW9kZWxGb3JPYmplY3REZXRlY3Rpb24uZnJvbV9wcmV0cmFpbmVkKCUyMklERUEtUmVzZWFyY2glMkZkYWItZGV0ci1yZXNuZXQtNTAlMjIlMkMlMjBkZXZpY2VfbWFwJTNEJTIyYXV0byUyMiklMEElMEFpbnB1dHMlMjAlM0QlMjBpbWFnZV9wcm9jZXNzb3IoaW1hZ2VzJTNEaW1hZ2UlMkMlMjByZXR1cm5fdGVuc29ycyUzRCUyMnB0JTIyKS50byhtb2RlbC5kZXZpY2UpJTBBJTBBd2l0aCUyMHRvcmNoLm5vX2dyYWQoKSUzQSUwQSUyMCUyMCUyMCUyMG91dHB1dHMlMjAlM0QlMjBtb2RlbCgqKmlucHV0cyklMEElMEFyZXN1bHRzJTIwJTNEJTIwaW1hZ2VfcHJvY2Vzc29yLnBvc3RfcHJvY2Vzc19vYmplY3RfZGV0ZWN0aW9uKG91dHB1dHMlMkMlMjB0YXJnZXRfc2l6ZXMlM0R0b3JjaC50ZW5zb3IoJTVCaW1hZ2Uuc2l6ZSU1QiUzQSUzQS0xJTVEJTVEKSUyQyUyMHRocmVzaG9sZCUzRDAuMyklMEElMEFmb3IlMjByZXN1bHQlMjBpbiUyMHJlc3VsdHMlM0ElMEElMjAlMjAlMjAlMjBmb3IlMjBzY29yZSUyQyUyMGxhYmVsX2lkJTJDJTIwYm94JTIwaW4lMjB6aXAocmVzdWx0JTVCJTIyc2NvcmVzJTIyJTVEJTJDJTIwcmVzdWx0JTVCJTIybGFiZWxzJTIyJTVEJTJDJTIwcmVzdWx0JTVCJTIyYm94ZXMlMjIlNUQpJTNBJTBBJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwc2NvcmUlMkMlMjBsYWJlbCUyMCUzRCUyMHNjb3JlLml0ZW0oKSUyQyUyMGxhYmVsX2lkLml0ZW0oKSUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMGJveCUyMCUzRCUyMCU1QnJvdW5kKGklMkMlMjAyKSUyMGZvciUyMGklMjBpbiUyMGJveC50b2xpc3QoKSU1RCUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMHByaW50KGYlMjIlN0Jtb2RlbC5jb25maWcuaWQybGFiZWwlNUJsYWJlbCU1RCU3RCUzQSUyMCU3QnNjb3JlJTNBLjJmJTdEJTIwJTdCYm94JTdEJTIyKQ==",highlighted:`<span class="hljs-keyword">import</span> requests
<span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> PIL <span class="hljs-keyword">import</span> Image
<span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoImageProcessor, AutoModelForObjectDetection
url = <span class="hljs-string">&#x27;http://images.cocodataset.org/val2017/000000039769.jpg&#x27;</span>
image = Image.<span class="hljs-built_in">open</span>(requests.get(url, stream=<span class="hljs-literal">True</span>).raw)
image_processor = AutoImageProcessor.from_pretrained(<span class="hljs-string">&quot;IDEA-Research/dab-detr-resnet-50&quot;</span>)
model = AutoModelForObjectDetection.from_pretrained(<span class="hljs-string">&quot;IDEA-Research/dab-detr-resnet-50&quot;</span>, device_map=<span class="hljs-string">&quot;auto&quot;</span>)
inputs = image_processor(images=image, return_tensors=<span class="hljs-string">&quot;pt&quot;</span>).to(model.device)
<span class="hljs-keyword">with</span> torch.no_grad():
outputs = model(**inputs)
results = image_processor.post_process_object_detection(outputs, target_sizes=torch.tensor([image.size[::-<span class="hljs-number">1</span>]]), threshold=<span class="hljs-number">0.3</span>)
<span class="hljs-keyword">for</span> result <span class="hljs-keyword">in</span> results:
<span class="hljs-keyword">for</span> score, label_id, box <span class="hljs-keyword">in</span> <span class="hljs-built_in">zip</span>(result[<span class="hljs-string">&quot;scores&quot;</span>], result[<span class="hljs-string">&quot;labels&quot;</span>], result[<span class="hljs-string">&quot;boxes&quot;</span>]):
score, label = score.item(), label_id.item()
box = [<span class="hljs-built_in">round</span>(i, <span class="hljs-number">2</span>) <span class="hljs-keyword">for</span> i <span class="hljs-keyword">in</span> box.tolist()]
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;<span class="hljs-subst">{model.config.id2label[label]}</span>: <span class="hljs-subst">{score:<span class="hljs-number">.2</span>f}</span> <span class="hljs-subst">{box}</span>&quot;</span>)`,lang:"python",wrap:!1}}),K=new V({props:{code:"Y2F0JTNBJTIwMC44NyUyMCU1QjE0LjclMkMlMjA0OS4zOSUyQyUyMDMyMC41MiUyQyUyMDQ2OS4yOCU1RCUwQXJlbW90ZSUzQSUyMDAuODYlMjAlNUI0MS4wOCUyQyUyMDcyLjM3JTJDJTIwMTczLjM5JTJDJTIwMTE3LjIlNUQlMEFjYXQlM0ElMjAwLjg2JTIwJTVCMzQ0LjQ1JTJDJTIwMTkuNDMlMkMlMjA2MzkuODUlMkMlMjAzNjcuODYlNUQlMEFyZW1vdGUlM0ElMjAwLjYxJTIwJTVCMzM0LjI3JTJDJTIwNzUuOTMlMkMlMjAzNjcuOTIlMkMlMjAxODguODElNUQlMEFjb3VjaCUzQSUyMDAuNTklMjAlNUItMC4wNCUyQyUyMDEuMzQlMkMlMjA2MzkuOSUyQyUyMDQ3Ny4wOSU1RA==",highlighted:`cat: 0.87 [14.7, 49.39, 320.52, 469.28]
remote: 0.86 [41.08, 72.37, 173.39, 117.2]
cat: 0.86 [344.45, 19.43, 639.85, 367.86]
remote: 0.61 [334.27, 75.93, 367.92, 188.81]
couch: 0.59 [-0.04, 1.34, 639.9, 477.09]`,lang:"text",wrap:!1}}),oe=new V({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMERhYkRldHJGb3JPYmplY3REZXRlY3Rpb24lMEElMEElMEFtb2RlbCUyMCUzRCUyMERhYkRldHJGb3JPYmplY3REZXRlY3Rpb24uZnJvbV9wcmV0cmFpbmVkKCUyMklERUEtUmVzZWFyY2glMkZkYWItZGV0ci1yZXNuZXQtNTAlMjIlMkMlMjBkZXZpY2VfbWFwJTNEJTIyYXV0byUyMik=",highlighted:`<span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> DabDetrForObjectDetection
model = DabDetrForObjectDetection.from_pretrained(<span class="hljs-string">&quot;IDEA-Research/dab-detr-resnet-50&quot;</span>, device_map=<span class="hljs-string">&quot;auto&quot;</span>)`,lang:"python",wrap:!1}}),se=new V({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMERhYkRldHJDb25maWclMkMlMjBEYWJEZXRyRm9yT2JqZWN0RGV0ZWN0aW9uJTBBJTBBJTBBY29uZmlnJTIwJTNEJTIwRGFiRGV0ckNvbmZpZygpJTBBbW9kZWwlMjAlM0QlMjBEYWJEZXRyRm9yT2JqZWN0RGV0ZWN0aW9uKGNvbmZpZyk=",highlighted:`<span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> DabDetrConfig, DabDetrForObjectDetection
config = DabDetrConfig()
model = DabDetrForObjectDetection(config)`,lang:"python",wrap:!1}}),re=new V({props:{code:"Y29uZmlnJTIwJTNEJTIwRGFiRGV0ckNvbmZpZygpJTBBbW9kZWwlMjAlM0QlMjBEYWJEZXRyRm9yT2JqZWN0RGV0ZWN0aW9uKGNvbmZpZyk=",highlighted:`config = DabDetrConfig()
model = DabDetrForObjectDetection(config)`,lang:"py",wrap:!1}}),le=new xe({props:{title:"DabDetrConfig",local:"transformers.DabDetrConfig",headingTag:"h2"}}),ie=new $e({props:{name:"class transformers.DabDetrConfig",anchor:"transformers.DabDetrConfig",parameters:[{name:"transformers_version",val:": str | None = None"},{name:"architectures",val:": list[str] | None = None"},{name:"output_hidden_states",val:": bool | None = False"},{name:"return_dict",val:": bool | None = True"},{name:"dtype",val:": typing.Union[str, ForwardRef('torch.dtype'), NoneType] = None"},{name:"chunk_size_feed_forward",val:": int = 0"},{name:"id2label",val:": dict[int, str] | dict[str, str] | None = None"},{name:"label2id",val:": dict[str, int] | dict[str, str] | None = None"},{name:"problem_type",val:": typing.Optional[typing.Literal['regression', 'single_label_classification', 'multi_label_classification']] = None"},{name:"is_encoder_decoder",val:": bool = True"},{name:"backbone_config",val:": dict | transformers.configuration_utils.PreTrainedConfig | None = None"},{name:"num_queries",val:": int = 300"},{name:"encoder_layers",val:": int = 6"},{name:"encoder_ffn_dim",val:": int = 2048"},{name:"encoder_attention_heads",val:": int = 8"},{name:"decoder_layers",val:": int = 6"},{name:"decoder_ffn_dim",val:": int = 2048"},{name:"decoder_attention_heads",val:": int = 8"},{name:"activation_function",val:": str = 'prelu'"},{name:"hidden_size",val:": int = 256"},{name:"dropout",val:": float | int = 0.1"},{name:"attention_dropout",val:": float | int = 0.0"},{name:"activation_dropout",val:": float | int = 0.0"},{name:"init_std",val:": float = 0.02"},{name:"init_xavier_std",val:": float = 1.0"},{name:"auxiliary_loss",val:": bool = False"},{name:"dilation",val:": bool = False"},{name:"class_cost",val:": int = 2"},{name:"bbox_cost",val:": int = 5"},{name:"giou_cost",val:": int = 2"},{name:"cls_loss_coefficient",val:": int = 2"},{name:"bbox_loss_coefficient",val:": int = 5"},{name:"giou_loss_coefficient",val:": int = 2"},{name:"focal_alpha",val:": float = 0.25"},{name:"temperature_height",val:": int = 20"},{name:"temperature_width",val:": int = 20"},{name:"query_dim",val:": int = 4"},{name:"random_refpoints_xy",val:": bool = False"},{name:"keep_query_pos",val:": bool = False"},{name:"num_patterns",val:": int = 0"},{name:"normalize_before",val:": bool = False"},{name:"sine_position_embedding_scale",val:": float | None = None"},{name:"initializer_bias_prior_prob",val:": float | None = None"},{name:"tie_word_embeddings",val:": bool = True"}],parametersDescription:[{anchor:"transformers.DabDetrConfig.is_encoder_decoder",description:`<strong>is_encoder_decoder</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) &#x2014;
Whether the model is used as an encoder/decoder or not.`,name:"is_encoder_decoder"},{anchor:"transformers.DabDetrConfig.backbone_config",description:`<strong>backbone_config</strong> (<code>Union[dict, ~configuration_utils.PreTrainedConfig]</code>, <em>optional</em>) &#x2014;
The configuration of the backbone model.`,name:"backbone_config"},{anchor:"transformers.DabDetrConfig.num_queries",description:`<strong>num_queries</strong> (<code>int</code>, <em>optional</em>, defaults to 300) &#x2014;
Number of object queries, i.e. detection slots. This is the maximal number of objects
<a href="/docs/transformers/pr_41116/en/model_doc/dab-detr#transformers.DabDetrModel">DabDetrModel</a> can detect in a single image. For COCO, we recommend 100 queries.`,name:"num_queries"},{anchor:"transformers.DabDetrConfig.encoder_layers",description:`<strong>encoder_layers</strong> (<code>int</code>, <em>optional</em>, defaults to <code>6</code>) &#x2014;
Number of hidden layers in the Transformer encoder. Will use the same value as <code>num_layers</code> if not set.`,name:"encoder_layers"},{anchor:"transformers.DabDetrConfig.encoder_ffn_dim",description:`<strong>encoder_ffn_dim</strong> (<code>int</code>, <em>optional</em>, defaults to <code>2048</code>) &#x2014;
Dimensionality of the &#x201C;intermediate&#x201D; (often named feed-forward) layer in encoder.`,name:"encoder_ffn_dim"},{anchor:"transformers.DabDetrConfig.encoder_attention_heads",description:`<strong>encoder_attention_heads</strong> (<code>int</code>, <em>optional</em>, defaults to <code>8</code>) &#x2014;
Number of attention heads for each attention layer in the Transformer encoder.`,name:"encoder_attention_heads"},{anchor:"transformers.DabDetrConfig.decoder_layers",description:`<strong>decoder_layers</strong> (<code>int</code>, <em>optional</em>, defaults to <code>6</code>) &#x2014;
Number of hidden layers in the Transformer decoder. Will use the same value as <code>num_layers</code> if not set.`,name:"decoder_layers"},{anchor:"transformers.DabDetrConfig.decoder_ffn_dim",description:`<strong>decoder_ffn_dim</strong> (<code>int</code>, <em>optional</em>, defaults to <code>2048</code>) &#x2014;
Dimensionality of the &#x201C;intermediate&#x201D; (often named feed-forward) layer in decoder.`,name:"decoder_ffn_dim"},{anchor:"transformers.DabDetrConfig.decoder_attention_heads",description:`<strong>decoder_attention_heads</strong> (<code>int</code>, <em>optional</em>, defaults to <code>8</code>) &#x2014;
Number of attention heads for each attention layer in the Transformer decoder.`,name:"decoder_attention_heads"},{anchor:"transformers.DabDetrConfig.activation_function",description:`<strong>activation_function</strong> (<code>str</code>, <em>optional</em>, defaults to <code>prelu</code>) &#x2014;
The non-linear activation function (function or string) in the decoder. For example, <code>&quot;gelu&quot;</code>,
<code>&quot;relu&quot;</code>, <code>&quot;silu&quot;</code>, etc.`,name:"activation_function"},{anchor:"transformers.DabDetrConfig.hidden_size",description:`<strong>hidden_size</strong> (<code>int</code>, <em>optional</em>, defaults to <code>256</code>) &#x2014;
Dimension of the hidden representations.`,name:"hidden_size"},{anchor:"transformers.DabDetrConfig.dropout",description:`<strong>dropout</strong> (<code>Union[float, int]</code>, <em>optional</em>, defaults to <code>0.1</code>) &#x2014;
The ratio for all dropout layers.`,name:"dropout"},{anchor:"transformers.DabDetrConfig.attention_dropout",description:`<strong>attention_dropout</strong> (<code>Union[float, int]</code>, <em>optional</em>, defaults to <code>0.0</code>) &#x2014;
The dropout ratio for the attention probabilities.`,name:"attention_dropout"},{anchor:"transformers.DabDetrConfig.activation_dropout",description:`<strong>activation_dropout</strong> (<code>Union[float, int]</code>, <em>optional</em>, defaults to <code>0.0</code>) &#x2014;
The dropout ratio for activations inside the fully connected layer.`,name:"activation_dropout"},{anchor:"transformers.DabDetrConfig.init_std",description:`<strong>init_std</strong> (<code>float</code>, <em>optional</em>, defaults to <code>0.02</code>) &#x2014;
The standard deviation of the truncated_normal_initializer for initializing all weight matrices.`,name:"init_std"},{anchor:"transformers.DabDetrConfig.init_xavier_std",description:`<strong>init_xavier_std</strong> (<code>float</code>, <em>optional</em>, defaults to <code>1.0</code>) &#x2014;
The scaling factor used for the Xavier initialization of the cross-attention weights.`,name:"init_xavier_std"},{anchor:"transformers.DabDetrConfig.auxiliary_loss",description:`<strong>auxiliary_loss</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) &#x2014;
Whether auxiliary decoding losses (losses at each decoder layer) are to be used.`,name:"auxiliary_loss"},{anchor:"transformers.DabDetrConfig.dilation",description:`<strong>dilation</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) &#x2014;
Whether to replace stride with dilation in the last convolutional block (DC5). Only supported when <code>use_timm_backbone</code> = <code>True</code>.`,name:"dilation"},{anchor:"transformers.DabDetrConfig.class_cost",description:`<strong>class_cost</strong> (<code>int</code>, <em>optional</em>, defaults to <code>2</code>) &#x2014;
Relative weight of the classification error in the Hungarian matching cost.`,name:"class_cost"},{anchor:"transformers.DabDetrConfig.bbox_cost",description:`<strong>bbox_cost</strong> (<code>int</code>, <em>optional</em>, defaults to <code>5</code>) &#x2014;
Relative weight of the L1 bounding box error in the Hungarian matching cost.`,name:"bbox_cost"},{anchor:"transformers.DabDetrConfig.giou_cost",description:`<strong>giou_cost</strong> (<code>int</code>, <em>optional</em>, defaults to <code>2</code>) &#x2014;
Relative weight of the generalized IoU loss in the Hungarian matching cost.`,name:"giou_cost"},{anchor:"transformers.DabDetrConfig.cls_loss_coefficient",description:`<strong>cls_loss_coefficient</strong> (<code>int</code>, <em>optional</em>, defaults to <code>2</code>) &#x2014;
Relative weight of the classification loss in the panoptic segmentation loss.`,name:"cls_loss_coefficient"},{anchor:"transformers.DabDetrConfig.bbox_loss_coefficient",description:`<strong>bbox_loss_coefficient</strong> (<code>int</code>, <em>optional</em>, defaults to <code>5</code>) &#x2014;
Relative weight of the L1 bounding box loss in the panoptic segmentation loss.`,name:"bbox_loss_coefficient"},{anchor:"transformers.DabDetrConfig.giou_loss_coefficient",description:`<strong>giou_loss_coefficient</strong> (<code>int</code>, <em>optional</em>, defaults to <code>2</code>) &#x2014;
Relative weight of the generalized IoU loss in the panoptic segmentation loss.`,name:"giou_loss_coefficient"},{anchor:"transformers.DabDetrConfig.focal_alpha",description:`<strong>focal_alpha</strong> (<code>float</code>, <em>optional</em>, defaults to <code>0.25</code>) &#x2014;
Alpha parameter in the focal loss.`,name:"focal_alpha"},{anchor:"transformers.DabDetrConfig.temperature_height",description:`<strong>temperature_height</strong> (<code>int</code>, <em>optional</em>, defaults to 20) &#x2014;
Temperature parameter to tune the flatness of positional attention (HEIGHT)`,name:"temperature_height"},{anchor:"transformers.DabDetrConfig.temperature_width",description:`<strong>temperature_width</strong> (<code>int</code>, <em>optional</em>, defaults to 20) &#x2014;
Temperature parameter to tune the flatness of positional attention (WIDTH)`,name:"temperature_width"},{anchor:"transformers.DabDetrConfig.query_dim",description:`<strong>query_dim</strong> (<code>int</code>, <em>optional</em>, defaults to 4) &#x2014;
Query dimension parameter represents the size of the output vector.`,name:"query_dim"},{anchor:"transformers.DabDetrConfig.random_refpoints_xy",description:`<strong>random_refpoints_xy</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) &#x2014;
Whether to fix the x and y coordinates of the anchor boxes with random initialization.`,name:"random_refpoints_xy"},{anchor:"transformers.DabDetrConfig.keep_query_pos",description:`<strong>keep_query_pos</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) &#x2014;
Whether to concatenate the projected positional embedding from the object query into the original query (key) in every decoder layer.`,name:"keep_query_pos"},{anchor:"transformers.DabDetrConfig.num_patterns",description:`<strong>num_patterns</strong> (<code>int</code>, <em>optional</em>, defaults to 0) &#x2014;
Number of pattern embeddings.`,name:"num_patterns"},{anchor:"transformers.DabDetrConfig.normalize_before",description:`<strong>normalize_before</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) &#x2014;
Whether we use a normalization layer in the Encoder or not.`,name:"normalize_before"},{anchor:"transformers.DabDetrConfig.sine_position_embedding_scale",description:`<strong>sine_position_embedding_scale</strong> (<code>float</code>, <em>optional</em>, defaults to &#x2018;None&#x2019;) &#x2014;
Scaling factor applied to the normalized positional encodings.`,name:"sine_position_embedding_scale"},{anchor:"transformers.DabDetrConfig.initializer_bias_prior_prob",description:`<strong>initializer_bias_prior_prob</strong> (<code>float</code>, <em>optional</em>) &#x2014;
The prior probability used by the bias initializer to initialize biases for <code>enc_score_head</code> and <code>class_embed</code>.
If <code>None</code>, <code>prior_prob</code> computed as <code>prior_prob = 1 / (num_labels + 1)</code> while initializing model weights.`,name:"initializer_bias_prior_prob"},{anchor:"transformers.DabDetrConfig.tie_word_embeddings",description:`<strong>tie_word_embeddings</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) &#x2014;
Whether to tie weight embeddings according to model&#x2019;s <code>tied_weights_keys</code> mapping.`,name:"tie_word_embeddings"}],source:"https://github.com/huggingface/transformers/blob/vr_41116/src/transformers/models/dab_detr/configuration_dab_detr.py#L26"}}),F=new Ut({props:{anchor:"transformers.DabDetrConfig.example",$$slots:{default:[ao]},$$scope:{ctx:Z}}}),de=new xe({props:{title:"DabDetrModel",local:"transformers.DabDetrModel",headingTag:"h2"}}),ce=new $e({props:{name:"class transformers.DabDetrModel",anchor:"transformers.DabDetrModel",parameters:[{name:"config",val:": DabDetrConfig"}],parametersDescription:[{anchor:"transformers.DabDetrModel.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_41116/en/model_doc/dab-detr#transformers.DabDetrConfig">DabDetrConfig</a>) &#x2014;
Model configuration class with all the parameters of the model. Initializing with a config file does not
load the weights associated with the model, only the configuration. Check out the
<a href="/docs/transformers/pr_41116/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_41116/src/transformers/models/dab_detr/modeling_dab_detr.py#L1155"}}),pe=new $e({props:{name:"forward",anchor:"transformers.DabDetrModel.forward",parameters:[{name:"pixel_values",val:": FloatTensor"},{name:"pixel_mask",val:": torch.LongTensor | None = None"},{name:"decoder_attention_mask",val:": torch.LongTensor | None = None"},{name:"encoder_outputs",val:": torch.FloatTensor | None = None"},{name:"inputs_embeds",val:": torch.FloatTensor | None = None"},{name:"decoder_inputs_embeds",val:": torch.FloatTensor | None = None"},{name:"output_attentions",val:": bool | None = None"},{name:"output_hidden_states",val:": bool | None = None"},{name:"return_dict",val:": bool | None = None"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.DabDetrModel.forward.pixel_values",description:`<strong>pixel_values</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_channels, image_size, image_size)</code>) &#x2014;
The tensors corresponding to the input images. Pixel values can be obtained using
<code>image_processor_class</code>. See <code>image_processor_class.__call__</code> for details (<code>processor_class</code> uses
<code>image_processor_class</code> for processing images).`,name:"pixel_values"},{anchor:"transformers.DabDetrModel.forward.pixel_mask",description:`<strong>pixel_mask</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, height, width)</code>, <em>optional</em>) &#x2014;
Mask to avoid performing attention on padding pixel values. Mask values selected in <code>[0, 1]</code>:</p>
<ul>
<li>1 for pixels that are real (i.e. <strong>not masked</strong>),</li>
<li>0 for pixels that are padding (i.e. <strong>masked</strong>).</li>
</ul>
<p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"pixel_mask"},{anchor:"transformers.DabDetrModel.forward.decoder_attention_mask",description:`<strong>decoder_attention_mask</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries)</code>, <em>optional</em>) &#x2014;
Not used by default. Can be used to mask object queries.`,name:"decoder_attention_mask"},{anchor:"transformers.DabDetrModel.forward.encoder_outputs",description:`<strong>encoder_outputs</strong> (<code>torch.FloatTensor</code>, <em>optional</em>) &#x2014;
Tuple consists of (<code>last_hidden_state</code>, <em>optional</em>: <code>hidden_states</code>, <em>optional</em>: <code>attentions</code>)
<code>last_hidden_state</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) is a sequence of
hidden-states at the output of the last layer of the encoder. Used in the cross-attention of the decoder.`,name:"encoder_outputs"},{anchor:"transformers.DabDetrModel.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) &#x2014;
Optionally, instead of passing the flattened feature map (output of the backbone + projection layer), you
can choose to directly pass a flattened representation of an image.`,name:"inputs_embeds"},{anchor:"transformers.DabDetrModel.forward.decoder_inputs_embeds",description:`<strong>decoder_inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, hidden_size)</code>, <em>optional</em>) &#x2014;
Optionally, instead of initializing the queries with a tensor of zeros, you can choose to directly pass an
embedded representation.`,name:"decoder_inputs_embeds"},{anchor:"transformers.DabDetrModel.forward.output_attentions",description:`<strong>output_attentions</strong> (<code>bool</code>, <em>optional</em>) &#x2014;
Whether or not to return the attentions tensors of all attention layers. See <code>attentions</code> under returned
tensors for more detail.`,name:"output_attentions"},{anchor:"transformers.DabDetrModel.forward.output_hidden_states",description:`<strong>output_hidden_states</strong> (<code>bool</code>, <em>optional</em>) &#x2014;
Whether or not to return the hidden states of all layers. See <code>hidden_states</code> under returned tensors for
more detail.`,name:"output_hidden_states"},{anchor:"transformers.DabDetrModel.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>) &#x2014;
Whether or not to return a <a href="/docs/transformers/pr_41116/en/main_classes/output#transformers.utils.ModelOutput">ModelOutput</a> instead of a plain tuple.`,name:"return_dict"}],source:"https://github.com/huggingface/transformers/blob/vr_41116/src/transformers/models/dab_detr/modeling_dab_detr.py#L1207",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script>
<p>A <code>DabDetrModelOutput</code> or a tuple of
<code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various
elements depending on the configuration (<a
href="/docs/transformers/pr_41116/en/model_doc/dab-detr#transformers.DabDetrConfig"
>DabDetrConfig</a>) and inputs.</p>
`,returnType:`<script context="module">export const metadata = 'undefined';<\/script>
<p><code>DabDetrModelOutput</code> or <code>tuple(torch.FloatTensor)</code></p>
`}}),W=new Lt({props:{$$slots:{default:[ro]},$$scope:{ctx:Z}}}),N=new Ut({props:{anchor:"transformers.DabDetrModel.forward.example",$$slots:{default:[lo]},$$scope:{ctx:Z}}}),me=new xe({props:{title:"DabDetrForObjectDetection",local:"transformers.DabDetrForObjectDetection",headingTag:"h2"}}),he=new $e({props:{name:"class transformers.DabDetrForObjectDetection",anchor:"transformers.DabDetrForObjectDetection",parameters:[{name:"config",val:": DabDetrConfig"}],parametersDescription:[{anchor:"transformers.DabDetrForObjectDetection.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_41116/en/model_doc/dab-detr#transformers.DabDetrConfig">DabDetrConfig</a>) &#x2014;
Model configuration class with all the parameters of the model. Initializing with a config file does not
load the weights associated with the model, only the configuration. Check out the
<a href="/docs/transformers/pr_41116/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_41116/src/transformers/models/dab_detr/modeling_dab_detr.py#L1427"}}),ue=new $e({props:{name:"forward",anchor:"transformers.DabDetrForObjectDetection.forward",parameters:[{name:"pixel_values",val:": FloatTensor"},{name:"pixel_mask",val:": torch.LongTensor | None = None"},{name:"decoder_attention_mask",val:": torch.LongTensor | None = None"},{name:"encoder_outputs",val:": torch.FloatTensor | None = None"},{name:"inputs_embeds",val:": torch.FloatTensor | None = None"},{name:"decoder_inputs_embeds",val:": torch.FloatTensor | None = None"},{name:"labels",val:": list[dict] | None = None"},{name:"output_attentions",val:": bool | None = None"},{name:"output_hidden_states",val:": bool | None = None"},{name:"return_dict",val:": bool | None = None"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.DabDetrForObjectDetection.forward.pixel_values",description:`<strong>pixel_values</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_channels, image_size, image_size)</code>) &#x2014;
The tensors corresponding to the input images. Pixel values can be obtained using
<code>image_processor_class</code>. See <code>image_processor_class.__call__</code> for details (<code>processor_class</code> uses
<code>image_processor_class</code> for processing images).`,name:"pixel_values"},{anchor:"transformers.DabDetrForObjectDetection.forward.pixel_mask",description:`<strong>pixel_mask</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, height, width)</code>, <em>optional</em>) &#x2014;
Mask to avoid performing attention on padding pixel values. Mask values selected in <code>[0, 1]</code>:</p>
<ul>
<li>1 for pixels that are real (i.e. <strong>not masked</strong>),</li>
<li>0 for pixels that are padding (i.e. <strong>masked</strong>).</li>
</ul>
<p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"pixel_mask"},{anchor:"transformers.DabDetrForObjectDetection.forward.decoder_attention_mask",description:`<strong>decoder_attention_mask</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries)</code>, <em>optional</em>) &#x2014;
Not used by default. Can be used to mask object queries.`,name:"decoder_attention_mask"},{anchor:"transformers.DabDetrForObjectDetection.forward.encoder_outputs",description:`<strong>encoder_outputs</strong> (<code>torch.FloatTensor</code>, <em>optional</em>) &#x2014;
Tuple consists of (<code>last_hidden_state</code>, <em>optional</em>: <code>hidden_states</code>, <em>optional</em>: <code>attentions</code>)
<code>last_hidden_state</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) is a sequence of
hidden-states at the output of the last layer of the encoder. Used in the cross-attention of the decoder.`,name:"encoder_outputs"},{anchor:"transformers.DabDetrForObjectDetection.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) &#x2014;
Optionally, instead of passing the flattened feature map (output of the backbone + projection layer), you
can choose to directly pass a flattened representation of an image.`,name:"inputs_embeds"},{anchor:"transformers.DabDetrForObjectDetection.forward.decoder_inputs_embeds",description:`<strong>decoder_inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, hidden_size)</code>, <em>optional</em>) &#x2014;
Optionally, instead of initializing the queries with a tensor of zeros, you can choose to directly pass an
embedded representation.`,name:"decoder_inputs_embeds"},{anchor:"transformers.DabDetrForObjectDetection.forward.labels",description:`<strong>labels</strong> (<code>list[Dict]</code> of len <code>(batch_size,)</code>, <em>optional</em>) &#x2014;
Labels for computing the bipartite matching loss. List of dicts, each dictionary containing at least the
following 2 keys: &#x2018;class_labels&#x2019; and &#x2018;boxes&#x2019; (the class labels and bounding boxes of an image in the batch
respectively). The class labels themselves should be a <code>torch.LongTensor</code> of len <code>(number of bounding boxes in the image,)</code> and the boxes a <code>torch.FloatTensor</code> of shape <code>(number of bounding boxes in the image, 4)</code>.`,name:"labels"},{anchor:"transformers.DabDetrForObjectDetection.forward.output_attentions",description:`<strong>output_attentions</strong> (<code>bool</code>, <em>optional</em>) &#x2014;
Whether or not to return the attentions tensors of all attention layers. See <code>attentions</code> under returned
tensors for more detail.`,name:"output_attentions"},{anchor:"transformers.DabDetrForObjectDetection.forward.output_hidden_states",description:`<strong>output_hidden_states</strong> (<code>bool</code>, <em>optional</em>) &#x2014;
Whether or not to return the hidden states of all layers. See <code>hidden_states</code> under returned tensors for
more detail.`,name:"output_hidden_states"},{anchor:"transformers.DabDetrForObjectDetection.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>) &#x2014;
Whether or not to return a <a href="/docs/transformers/pr_41116/en/main_classes/output#transformers.utils.ModelOutput">ModelOutput</a> instead of a plain tuple.`,name:"return_dict"}],source:"https://github.com/huggingface/transformers/blob/vr_41116/src/transformers/models/dab_detr/modeling_dab_detr.py#L1456",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script>
<p>A <code>DabDetrObjectDetectionOutput</code> or a tuple of
<code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various
elements depending on the configuration (<a
href="/docs/transformers/pr_41116/en/model_doc/dab-detr#transformers.DabDetrConfig"
>DabDetrConfig</a>) and inputs.</p>
`,returnType:`<script context="module">export const metadata = 'undefined';<\/script>
<p><code>DabDetrObjectDetectionOutput</code> or <code>tuple(torch.FloatTensor)</code></p>
`}}),B=new Lt({props:{$$slots:{default:[io]},$$scope:{ctx:Z}}}),E=new Ut({props:{anchor:"transformers.DabDetrForObjectDetection.forward.example",$$slots:{default:[co]},$$scope:{ctx:Z}}}),fe=new so({props:{source:"https://github.com/huggingface/transformers/blob/main/docs/source/en/model_doc/dab-detr.md"}}),{c(){r=i("meta"),w=s(),p=i("p"),m=s(),M=i("p"),M.innerHTML=l,T=s(),u(G.$$.fragment),ze=s(),u(q.$$.fragment),Re=s(),u(Q.$$.fragment),Ie=s(),X=i("p"),X.innerHTML=Ct,Fe=s(),I=i("img"),We=s(),H=i("p"),H.textContent=xt,Ne=s(),A=i("p"),A.innerHTML=kt,Be=s(),S=i("p"),S.innerHTML=$t,Ee=s(),u(Y.$$.fragment),Ve=s(),L=i("p"),L.textContent=Zt,Ge=s(),u(O.$$.fragment),qe=s(),P=i("p"),P.textContent=zt,Qe=s(),u(K.$$.fragment),Xe=s(),ee=i("p"),ee.textContent=Rt,He=s(),te=i("p"),te.textContent=It,Ae=s(),u(oe.$$.fragment),Se=s(),ne=i("p"),ne.textContent=Ft,Ye=s(),u(se.$$.fragment),Le=s(),ae=i("p"),ae.textContent=Wt,Oe=s(),u(re.$$.fragment),Pe=s(),u(le.$$.fragment),Ke=s(),C=i("div"),u(ie.$$.fragment),lt=s(),be=i("p"),be.innerHTML=Nt,it=s(),_e=i("p"),_e.innerHTML=Bt,dt=s(),u(F.$$.fragment),et=s(),u(de.$$.fragment),tt=s(),j=i("div"),u(ce.$$.fragment),ct=s(),ye=i("p"),ye.textContent=Et,pt=s(),Me=i("p"),Me.innerHTML=Vt,mt=s(),we=i("p"),we.innerHTML=Gt,ht=s(),v=i("div"),u(pe.$$.fragment),ut=s(),Te=i("p"),Te.innerHTML=qt,ft=s(),u(W.$$.fragment),gt=s(),je=i("ul"),je.innerHTML=Qt,bt=s(),u(N.$$.fragment),ot=s(),u(me.$$.fragment),nt=s(),D=i("div"),u(he.$$.fragment),_t=s(),De=i("p"),De.textContent=Xt,yt=s(),ve=i("p"),ve.innerHTML=Ht,Mt=s(),Ue=i("p"),Ue.innerHTML=At,wt=s(),U=i("div"),u(ue.$$.fragment),Tt=s(),Ce=i("p"),Ce.innerHTML=St,jt=s(),u(B.$$.fragment),Dt=s(),Je=i("ul"),Je.innerHTML=Yt,vt=s(),u(E.$$.fragment),st=s(),u(fe.$$.fragment),at=s(),ke=i("p"),this.h()},l(e){const t=oo("svelte-u9bgzb",document.head);r=d(t,"META",{name:!0,content:!0}),t.forEach(o),w=a(e),p=d(e,"P",{}),ge(p).forEach(o),m=a(e),M=d(e,"P",{"data-svelte-h":!0}),h(M)!=="svelte-xplwdc"&&(M.innerHTML=l),T=a(e),f(G.$$.fragment,e),ze=a(e),f(q.$$.fragment,e),Re=a(e),f(Q.$$.fragment,e),Ie=a(e),X=d(e,"P",{"data-svelte-h":!0}),h(X)!=="svelte-1a1lcm7"&&(X.innerHTML=Ct),Fe=a(e),I=d(e,"IMG",{src:!0,alt:!0,width:!0}),We=a(e),H=d(e,"P",{"data-svelte-h":!0}),h(H)!=="svelte-vfdo9a"&&(H.textContent=xt),Ne=a(e),A=d(e,"P",{"data-svelte-h":!0}),h(A)!=="svelte-ct3tl8"&&(A.innerHTML=kt),Be=a(e),S=d(e,"P",{"data-svelte-h":!0}),h(S)!=="svelte-yd6iz"&&(S.innerHTML=$t),Ee=a(e),f(Y.$$.fragment,e),Ve=a(e),L=d(e,"P",{"data-svelte-h":!0}),h(L)!=="svelte-15ft1fk"&&(L.textContent=Zt),Ge=a(e),f(O.$$.fragment,e),qe=a(e),P=d(e,"P",{"data-svelte-h":!0}),h(P)!=="svelte-192r0ao"&&(P.textContent=zt),Qe=a(e),f(K.$$.fragment,e),Xe=a(e),ee=d(e,"P",{"data-svelte-h":!0}),h(ee)!=="svelte-1bp0kxx"&&(ee.textContent=Rt),He=a(e),te=d(e,"P",{"data-svelte-h":!0}),h(te)!=="svelte-1m4ojtg"&&(te.textContent=It),Ae=a(e),f(oe.$$.fragment,e),Se=a(e),ne=d(e,"P",{"data-svelte-h":!0}),h(ne)!=="svelte-fa3upe"&&(ne.textContent=Ft),Ye=a(e),f(se.$$.fragment,e),Le=a(e),ae=d(e,"P",{"data-svelte-h":!0}),h(ae)!=="svelte-1m4gpgg"&&(ae.textContent=Wt),Oe=a(e),f(re.$$.fragment,e),Pe=a(e),f(le.$$.fragment,e),Ke=a(e),C=d(e,"DIV",{class:!0});var z=ge(C);f(ie.$$.fragment,z),lt=a(z),be=d(z,"P",{"data-svelte-h":!0}),h(be)!=="svelte-1tmkbf2"&&(be.innerHTML=Nt),it=a(z),_e=d(z,"P",{"data-svelte-h":!0}),h(_e)!=="svelte-1plkghn"&&(_e.innerHTML=Bt),dt=a(z),f(F.$$.fragment,z),z.forEach(o),et=a(e),f(de.$$.fragment,e),tt=a(e),j=d(e,"DIV",{class:!0});var J=ge(j);f(ce.$$.fragment,J),ct=a(J),ye=d(J,"P",{"data-svelte-h":!0}),h(ye)!=="svelte-19gfwdg"&&(ye.textContent=Et),pt=a(J),Me=d(J,"P",{"data-svelte-h":!0}),h(Me)!=="svelte-g65tf5"&&(Me.innerHTML=Vt),mt=a(J),we=d(J,"P",{"data-svelte-h":!0}),h(we)!=="svelte-hswkmf"&&(we.innerHTML=Gt),ht=a(J),v=d(J,"DIV",{class:!0});var x=ge(v);f(pe.$$.fragment,x),ut=a(x),Te=d(x,"P",{"data-svelte-h":!0}),h(Te)!=="svelte-jrsq4j"&&(Te.innerHTML=qt),ft=a(x),f(W.$$.fragment,x),gt=a(x),je=d(x,"UL",{"data-svelte-h":!0}),h(je)!=="svelte-176abyt"&&(je.innerHTML=Qt),bt=a(x),f(N.$$.fragment,x),x.forEach(o),J.forEach(o),ot=a(e),f(me.$$.fragment,e),nt=a(e),D=d(e,"DIV",{class:!0});var k=ge(D);f(he.$$.fragment,k),_t=a(k),De=d(k,"P",{"data-svelte-h":!0}),h(De)!=="svelte-13cvp2k"&&(De.textContent=Xt),yt=a(k),ve=d(k,"P",{"data-svelte-h":!0}),h(ve)!=="svelte-g65tf5"&&(ve.innerHTML=Ht),Mt=a(k),Ue=d(k,"P",{"data-svelte-h":!0}),h(Ue)!=="svelte-hswkmf"&&(Ue.innerHTML=At),wt=a(k),U=d(k,"DIV",{class:!0});var $=ge(U);f(ue.$$.fragment,$),Tt=a($),Ce=d($,"P",{"data-svelte-h":!0}),h(Ce)!=="svelte-mifrp"&&(Ce.innerHTML=St),jt=a($),f(B.$$.fragment,$),Dt=a($),Je=d($,"UL",{"data-svelte-h":!0}),h(Je)!=="svelte-dr1j9z"&&(Je.innerHTML=Yt),vt=a($),f(E.$$.fragment,$),$.forEach(o),k.forEach(o),st=a(e),f(fe.$$.fragment,e),at=a(e),ke=d(e,"P",{}),ge(ke).forEach(o),this.h()},h(){R(r,"name","hf:doc:metadata"),R(r,"content",mo),Pt(I.src,Jt="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/transformers/model_doc/dab_detr_convergence_plot.png")||R(I,"src",Jt),R(I,"alt","drawing"),R(I,"width","600"),R(C,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),R(v,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),R(j,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),R(U,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),R(D,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8")},m(e,t){c(document.head,r),n(e,w,t),n(e,p,t),n(e,m,t),n(e,M,t),n(e,T,t),g(G,e,t),n(e,ze,t),g(q,e,t),n(e,Re,t),g(Q,e,t),n(e,Ie,t),n(e,X,t),n(e,Fe,t),n(e,I,t),n(e,We,t),n(e,H,t),n(e,Ne,t),n(e,A,t),n(e,Be,t),n(e,S,t),n(e,Ee,t),g(Y,e,t),n(e,Ve,t),n(e,L,t),n(e,Ge,t),g(O,e,t),n(e,qe,t),n(e,P,t),n(e,Qe,t),g(K,e,t),n(e,Xe,t),n(e,ee,t),n(e,He,t),n(e,te,t),n(e,Ae,t),g(oe,e,t),n(e,Se,t),n(e,ne,t),n(e,Ye,t),g(se,e,t),n(e,Le,t),n(e,ae,t),n(e,Oe,t),g(re,e,t),n(e,Pe,t),g(le,e,t),n(e,Ke,t),n(e,C,t),g(ie,C,null),c(C,lt),c(C,be),c(C,it),c(C,_e),c(C,dt),g(F,C,null),n(e,et,t),g(de,e,t),n(e,tt,t),n(e,j,t),g(ce,j,null),c(j,ct),c(j,ye),c(j,pt),c(j,Me),c(j,mt),c(j,we),c(j,ht),c(j,v),g(pe,v,null),c(v,ut),c(v,Te),c(v,ft),g(W,v,null),c(v,gt),c(v,je),c(v,bt),g(N,v,null),n(e,ot,t),g(me,e,t),n(e,nt,t),n(e,D,t),g(he,D,null),c(D,_t),c(D,De),c(D,yt),c(D,ve),c(D,Mt),c(D,Ue),c(D,wt),c(D,U),g(ue,U,null),c(U,Tt),c(U,Ce),c(U,jt),g(B,U,null),c(U,Dt),c(U,Je),c(U,vt),g(E,U,null),n(e,st,t),g(fe,e,t),n(e,at,t),n(e,ke,t),rt=!0},p(e,[t]){const z={};t&2&&(z.$$scope={dirty:t,ctx:e}),F.$set(z);const J={};t&2&&(J.$$scope={dirty:t,ctx:e}),W.$set(J);const x={};t&2&&(x.$$scope={dirty:t,ctx:e}),N.$set(x);const k={};t&2&&(k.$$scope={dirty:t,ctx:e}),B.$set(k);const $={};t&2&&($.$$scope={dirty:t,ctx:e}),E.$set($)},i(e){rt||(b(G.$$.fragment,e),b(q.$$.fragment,e),b(Q.$$.fragment,e),b(Y.$$.fragment,e),b(O.$$.fragment,e),b(K.$$.fragment,e),b(oe.$$.fragment,e),b(se.$$.fragment,e),b(re.$$.fragment,e),b(le.$$.fragment,e),b(ie.$$.fragment,e),b(F.$$.fragment,e),b(de.$$.fragment,e),b(ce.$$.fragment,e),b(pe.$$.fragment,e),b(W.$$.fragment,e),b(N.$$.fragment,e),b(me.$$.fragment,e),b(he.$$.fragment,e),b(ue.$$.fragment,e),b(B.$$.fragment,e),b(E.$$.fragment,e),b(fe.$$.fragment,e),rt=!0)},o(e){_(G.$$.fragment,e),_(q.$$.fragment,e),_(Q.$$.fragment,e),_(Y.$$.fragment,e),_(O.$$.fragment,e),_(K.$$.fragment,e),_(oe.$$.fragment,e),_(se.$$.fragment,e),_(re.$$.fragment,e),_(le.$$.fragment,e),_(ie.$$.fragment,e),_(F.$$.fragment,e),_(de.$$.fragment,e),_(ce.$$.fragment,e),_(pe.$$.fragment,e),_(W.$$.fragment,e),_(N.$$.fragment,e),_(me.$$.fragment,e),_(he.$$.fragment,e),_(ue.$$.fragment,e),_(B.$$.fragment,e),_(E.$$.fragment,e),_(fe.$$.fragment,e),rt=!1},d(e){e&&(o(w),o(p),o(m),o(M),o(T),o(ze),o(Re),o(Ie),o(X),o(Fe),o(I),o(We),o(H),o(Ne),o(A),o(Be),o(S),o(Ee),o(Ve),o(L),o(Ge),o(qe),o(P),o(Qe),o(Xe),o(ee),o(He),o(te),o(Ae),o(Se),o(ne),o(Ye),o(Le),o(ae),o(Oe),o(Pe),o(Ke),o(C),o(et),o(tt),o(j),o(ot),o(nt),o(D),o(st),o(at),o(ke)),o(r),y(G,e),y(q,e),y(Q,e),y(Y,e),y(O,e),y(K,e),y(oe,e),y(se,e),y(re,e),y(le,e),y(ie),y(F),y(de,e),y(ce),y(pe),y(W),y(N),y(me,e),y(he),y(ue),y(B),y(E),y(fe,e)}}}const mo='{"title":"DAB-DETR","local":"dab-detr","sections":[{"title":"Overview","local":"overview","sections":[],"depth":2},{"title":"How to Get Started with the Model","local":"how-to-get-started-with-the-model","sections":[],"depth":2},{"title":"DabDetrConfig","local":"transformers.DabDetrConfig","sections":[],"depth":2},{"title":"DabDetrModel","local":"transformers.DabDetrModel","sections":[],"depth":2},{"title":"DabDetrForObjectDetection","local":"transformers.DabDetrForObjectDetection","sections":[],"depth":2}],"depth":1}';function ho(Z){return Kt(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class To extends eo{constructor(r){super(),to(this,r,ho,po,Ot,{})}}export{To as component};

Xet Storage Details

Size:
71.6 kB
·
Xet hash:
44ad3da24e86e354e9c6cdc1b5c4328a334d2d48905c76cb003f43e4d55ca4cc

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.