Buckets:
| import{s as Sr,o as Wr,n as Lt}from"../chunks/scheduler.01eeda35.js";import{S as Hr,i as Br,g as a,s,r as p,A as Ar,h as i,f as t,c as r,j as v,u as h,x as m,k as T,y as n,a as d,v as g,d as f,t as u,w as _}from"../chunks/index.6dd51b66.js";import{T as Es}from"../chunks/Tip.de9bae2b.js";import{D as w}from"../chunks/Docstring.76e6b3cf.js";import{C as Jt}from"../chunks/CodeBlock.864da1b0.js";import{E as In}from"../chunks/ExampleCodeBlock.6a36fb6b.js";import{P as Gr}from"../chunks/PipelineTag.5efc345e.js";import{H as J,E as Vr}from"../chunks/EditOnGithub.7faefd25.js";function Xr(k){let c,x="Examples:",b,y,D;return y=new Jt({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMERldHJDb25maWclMkMlMjBEZXRyTW9kZWwlMEElMEElMjMlMjBJbml0aWFsaXppbmclMjBhJTIwREVUUiUyMGZhY2Vib29rJTJGZGV0ci1yZXNuZXQtNTAlMjBzdHlsZSUyMGNvbmZpZ3VyYXRpb24lMEFjb25maWd1cmF0aW9uJTIwJTNEJTIwRGV0ckNvbmZpZygpJTBBJTBBJTIzJTIwSW5pdGlhbGl6aW5nJTIwYSUyMG1vZGVsJTIwKHdpdGglMjByYW5kb20lMjB3ZWlnaHRzKSUyMGZyb20lMjB0aGUlMjBmYWNlYm9vayUyRmRldHItcmVzbmV0LTUwJTIwc3R5bGUlMjBjb25maWd1cmF0aW9uJTBBbW9kZWwlMjAlM0QlMjBEZXRyTW9kZWwoY29uZmlndXJhdGlvbiklMEElMEElMjMlMjBBY2Nlc3NpbmclMjB0aGUlMjBtb2RlbCUyMGNvbmZpZ3VyYXRpb24lMEFjb25maWd1cmF0aW9uJTIwJTNEJTIwbW9kZWwuY29uZmln",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> DetrConfig, DetrModel | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Initializing a DETR facebook/detr-resnet-50 style configuration</span> | |
| <span class="hljs-meta">>>> </span>configuration = DetrConfig() | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Initializing a model (with random weights) from the facebook/detr-resnet-50 style configuration</span> | |
| <span class="hljs-meta">>>> </span>model = DetrModel(configuration) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Accessing the model configuration</span> | |
| <span class="hljs-meta">>>> </span>configuration = model.config`,wrap:!1}}),{c(){c=a("p"),c.textContent=x,b=s(),p(y.$$.fragment)},l(l){c=i(l,"P",{"data-svelte-h":!0}),m(c)!=="svelte-kvfsh7"&&(c.textContent=x),b=r(l),h(y.$$.fragment,l)},m(l,M){d(l,c,M),d(l,b,M),g(y,l,M),D=!0},p:Lt,i(l){D||(f(y.$$.fragment,l),D=!0)},o(l){u(y.$$.fragment,l),D=!1},d(l){l&&(t(c),t(b)),_(y,l)}}}function Yr(k){let c,x=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){c=a("p"),c.innerHTML=x},l(b){c=i(b,"P",{"data-svelte-h":!0}),m(c)!=="svelte-fincs2"&&(c.innerHTML=x)},m(b,y){d(b,c,y)},p:Lt,d(b){b&&t(c)}}}function Qr(k){let c,x="Examples:",b,y,D;return y=new Jt({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9JbWFnZVByb2Nlc3NvciUyQyUyMERldHJNb2RlbCUwQWZyb20lMjBQSUwlMjBpbXBvcnQlMjBJbWFnZSUwQWltcG9ydCUyMHJlcXVlc3RzJTBBJTBBdXJsJTIwJTNEJTIwJTIyaHR0cCUzQSUyRiUyRmltYWdlcy5jb2NvZGF0YXNldC5vcmclMkZ2YWwyMDE3JTJGMDAwMDAwMDM5NzY5LmpwZyUyMiUwQWltYWdlJTIwJTNEJTIwSW1hZ2Uub3BlbihyZXF1ZXN0cy5nZXQodXJsJTJDJTIwc3RyZWFtJTNEVHJ1ZSkucmF3KSUwQSUwQWltYWdlX3Byb2Nlc3NvciUyMCUzRCUyMEF1dG9JbWFnZVByb2Nlc3Nvci5mcm9tX3ByZXRyYWluZWQoJTIyZmFjZWJvb2slMkZkZXRyLXJlc25ldC01MCUyMiklMEFtb2RlbCUyMCUzRCUyMERldHJNb2RlbC5mcm9tX3ByZXRyYWluZWQoJTIyZmFjZWJvb2slMkZkZXRyLXJlc25ldC01MCUyMiklMEElMEElMjMlMjBwcmVwYXJlJTIwaW1hZ2UlMjBmb3IlMjB0aGUlMjBtb2RlbCUwQWlucHV0cyUyMCUzRCUyMGltYWdlX3Byb2Nlc3NvcihpbWFnZXMlM0RpbWFnZSUyQyUyMHJldHVybl90ZW5zb3JzJTNEJTIycHQlMjIpJTBBJTBBJTIzJTIwZm9yd2FyZCUyMHBhc3MlMEFvdXRwdXRzJTIwJTNEJTIwbW9kZWwoKippbnB1dHMpJTBBJTBBJTIzJTIwdGhlJTIwbGFzdCUyMGhpZGRlbiUyMHN0YXRlcyUyMGFyZSUyMHRoZSUyMGZpbmFsJTIwcXVlcnklMjBlbWJlZGRpbmdzJTIwb2YlMjB0aGUlMjBUcmFuc2Zvcm1lciUyMGRlY29kZXIlMEElMjMlMjB0aGVzZSUyMGFyZSUyMG9mJTIwc2hhcGUlMjAoYmF0Y2hfc2l6ZSUyQyUyMG51bV9xdWVyaWVzJTJDJTIwaGlkZGVuX3NpemUpJTBBbGFzdF9oaWRkZW5fc3RhdGVzJTIwJTNEJTIwb3V0cHV0cy5sYXN0X2hpZGRlbl9zdGF0ZSUwQWxpc3QobGFzdF9oaWRkZW5fc3RhdGVzLnNoYXBlKQ==",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoImageProcessor, DetrModel | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> PIL <span class="hljs-keyword">import</span> Image | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> requests | |
| <span class="hljs-meta">>>> </span>url = <span class="hljs-string">"http://images.cocodataset.org/val2017/000000039769.jpg"</span> | |
| <span class="hljs-meta">>>> </span>image = Image.<span class="hljs-built_in">open</span>(requests.get(url, stream=<span class="hljs-literal">True</span>).raw) | |
| <span class="hljs-meta">>>> </span>image_processor = AutoImageProcessor.from_pretrained(<span class="hljs-string">"facebook/detr-resnet-50"</span>) | |
| <span class="hljs-meta">>>> </span>model = DetrModel.from_pretrained(<span class="hljs-string">"facebook/detr-resnet-50"</span>) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># prepare image for the model</span> | |
| <span class="hljs-meta">>>> </span>inputs = image_processor(images=image, return_tensors=<span class="hljs-string">"pt"</span>) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># forward pass</span> | |
| <span class="hljs-meta">>>> </span>outputs = model(**inputs) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># the last hidden states are the final query embeddings of the Transformer decoder</span> | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># these are of shape (batch_size, num_queries, hidden_size)</span> | |
| <span class="hljs-meta">>>> </span>last_hidden_states = outputs.last_hidden_state | |
| <span class="hljs-meta">>>> </span><span class="hljs-built_in">list</span>(last_hidden_states.shape) | |
| [<span class="hljs-number">1</span>, <span class="hljs-number">100</span>, <span class="hljs-number">256</span>]`,wrap:!1}}),{c(){c=a("p"),c.textContent=x,b=s(),p(y.$$.fragment)},l(l){c=i(l,"P",{"data-svelte-h":!0}),m(c)!=="svelte-kvfsh7"&&(c.textContent=x),b=r(l),h(y.$$.fragment,l)},m(l,M){d(l,c,M),d(l,b,M),g(y,l,M),D=!0},p:Lt,i(l){D||(f(y.$$.fragment,l),D=!0)},o(l){u(y.$$.fragment,l),D=!1},d(l){l&&(t(c),t(b)),_(y,l)}}}function Kr(k){let c,x=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){c=a("p"),c.innerHTML=x},l(b){c=i(b,"P",{"data-svelte-h":!0}),m(c)!=="svelte-fincs2"&&(c.innerHTML=x)},m(b,y){d(b,c,y)},p:Lt,d(b){b&&t(c)}}}function ea(k){let c,x="Examples:",b,y,D;return y=new Jt({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9JbWFnZVByb2Nlc3NvciUyQyUyMERldHJGb3JPYmplY3REZXRlY3Rpb24lMEFpbXBvcnQlMjB0b3JjaCUwQWZyb20lMjBQSUwlMjBpbXBvcnQlMjBJbWFnZSUwQWltcG9ydCUyMHJlcXVlc3RzJTBBJTBBdXJsJTIwJTNEJTIwJTIyaHR0cCUzQSUyRiUyRmltYWdlcy5jb2NvZGF0YXNldC5vcmclMkZ2YWwyMDE3JTJGMDAwMDAwMDM5NzY5LmpwZyUyMiUwQWltYWdlJTIwJTNEJTIwSW1hZ2Uub3BlbihyZXF1ZXN0cy5nZXQodXJsJTJDJTIwc3RyZWFtJTNEVHJ1ZSkucmF3KSUwQSUwQWltYWdlX3Byb2Nlc3NvciUyMCUzRCUyMEF1dG9JbWFnZVByb2Nlc3Nvci5mcm9tX3ByZXRyYWluZWQoJTIyZmFjZWJvb2slMkZkZXRyLXJlc25ldC01MCUyMiklMEFtb2RlbCUyMCUzRCUyMERldHJGb3JPYmplY3REZXRlY3Rpb24uZnJvbV9wcmV0cmFpbmVkKCUyMmZhY2Vib29rJTJGZGV0ci1yZXNuZXQtNTAlMjIpJTBBJTBBaW5wdXRzJTIwJTNEJTIwaW1hZ2VfcHJvY2Vzc29yKGltYWdlcyUzRGltYWdlJTJDJTIwcmV0dXJuX3RlbnNvcnMlM0QlMjJwdCUyMiklMEFvdXRwdXRzJTIwJTNEJTIwbW9kZWwoKippbnB1dHMpJTBBJTBBJTIzJTIwY29udmVydCUyMG91dHB1dHMlMjAoYm91bmRpbmclMjBib3hlcyUyMGFuZCUyMGNsYXNzJTIwbG9naXRzKSUyMHRvJTIwUGFzY2FsJTIwVk9DJTIwZm9ybWF0JTIwKHhtaW4lMkMlMjB5bWluJTJDJTIweG1heCUyQyUyMHltYXgpJTBBdGFyZ2V0X3NpemVzJTIwJTNEJTIwdG9yY2gudGVuc29yKCU1QmltYWdlLnNpemUlNUIlM0ElM0EtMSU1RCU1RCklMEFyZXN1bHRzJTIwJTNEJTIwaW1hZ2VfcHJvY2Vzc29yLnBvc3RfcHJvY2Vzc19vYmplY3RfZGV0ZWN0aW9uKG91dHB1dHMlMkMlMjB0aHJlc2hvbGQlM0QwLjklMkMlMjB0YXJnZXRfc2l6ZXMlM0R0YXJnZXRfc2l6ZXMpJTVCJTBBJTIwJTIwJTIwJTIwMCUwQSU1RCUwQSUwQWZvciUyMHNjb3JlJTJDJTIwbGFiZWwlMkMlMjBib3glMjBpbiUyMHppcChyZXN1bHRzJTVCJTIyc2NvcmVzJTIyJTVEJTJDJTIwcmVzdWx0cyU1QiUyMmxhYmVscyUyMiU1RCUyQyUyMHJlc3VsdHMlNUIlMjJib3hlcyUyMiU1RCklM0ElMEElMjAlMjAlMjAlMjBib3glMjAlM0QlMjAlNUJyb3VuZChpJTJDJTIwMiklMjBmb3IlMjBpJTIwaW4lMjBib3gudG9saXN0KCklNUQlMEElMjAlMjAlMjAlMjBwcmludCglMEElMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjBmJTIyRGV0ZWN0ZWQlMjAlN0Jtb2RlbC5jb25maWcuaWQybGFiZWwlNUJsYWJlbC5pdGVtKCklNUQlN0QlMjB3aXRoJTIwY29uZmlkZW5jZSUyMCUyMiUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMGYlMjIlN0Jyb3VuZChzY29yZS5pdGVtKCklMkMlMjAzKSU3RCUyMGF0JTIwbG9jYXRpb24lMjAlN0Jib3glN0QlMjIlMEElMjAlMjAlMjAlMjAp",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoImageProcessor, DetrForObjectDetection | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> PIL <span class="hljs-keyword">import</span> Image | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> requests | |
| <span class="hljs-meta">>>> </span>url = <span class="hljs-string">"http://images.cocodataset.org/val2017/000000039769.jpg"</span> | |
| <span class="hljs-meta">>>> </span>image = Image.<span class="hljs-built_in">open</span>(requests.get(url, stream=<span class="hljs-literal">True</span>).raw) | |
| <span class="hljs-meta">>>> </span>image_processor = AutoImageProcessor.from_pretrained(<span class="hljs-string">"facebook/detr-resnet-50"</span>) | |
| <span class="hljs-meta">>>> </span>model = DetrForObjectDetection.from_pretrained(<span class="hljs-string">"facebook/detr-resnet-50"</span>) | |
| <span class="hljs-meta">>>> </span>inputs = image_processor(images=image, return_tensors=<span class="hljs-string">"pt"</span>) | |
| <span class="hljs-meta">>>> </span>outputs = model(**inputs) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># convert outputs (bounding boxes and class logits) to Pascal VOC format (xmin, ymin, xmax, ymax)</span> | |
| <span class="hljs-meta">>>> </span>target_sizes = torch.tensor([image.size[::-<span class="hljs-number">1</span>]]) | |
| <span class="hljs-meta">>>> </span>results = image_processor.post_process_object_detection(outputs, threshold=<span class="hljs-number">0.9</span>, target_sizes=target_sizes)[ | |
| <span class="hljs-meta">... </span> <span class="hljs-number">0</span> | |
| <span class="hljs-meta">... </span>] | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">for</span> score, label, box <span class="hljs-keyword">in</span> <span class="hljs-built_in">zip</span>(results[<span class="hljs-string">"scores"</span>], results[<span class="hljs-string">"labels"</span>], results[<span class="hljs-string">"boxes"</span>]): | |
| <span class="hljs-meta">... </span> box = [<span class="hljs-built_in">round</span>(i, <span class="hljs-number">2</span>) <span class="hljs-keyword">for</span> i <span class="hljs-keyword">in</span> box.tolist()] | |
| <span class="hljs-meta">... </span> <span class="hljs-built_in">print</span>( | |
| <span class="hljs-meta">... </span> <span class="hljs-string">f"Detected <span class="hljs-subst">{model.config.id2label[label.item()]}</span> with confidence "</span> | |
| <span class="hljs-meta">... </span> <span class="hljs-string">f"<span class="hljs-subst">{<span class="hljs-built_in">round</span>(score.item(), <span class="hljs-number">3</span>)}</span> at location <span class="hljs-subst">{box}</span>"</span> | |
| <span class="hljs-meta">... </span> ) | |
| Detected remote <span class="hljs-keyword">with</span> confidence <span class="hljs-number">0.998</span> at location [<span class="hljs-number">40.16</span>, <span class="hljs-number">70.81</span>, <span class="hljs-number">175.55</span>, <span class="hljs-number">117.98</span>] | |
| Detected remote <span class="hljs-keyword">with</span> confidence <span class="hljs-number">0.996</span> at location [<span class="hljs-number">333.24</span>, <span class="hljs-number">72.55</span>, <span class="hljs-number">368.33</span>, <span class="hljs-number">187.66</span>] | |
| Detected couch <span class="hljs-keyword">with</span> confidence <span class="hljs-number">0.995</span> at location [-<span class="hljs-number">0.02</span>, <span class="hljs-number">1.15</span>, <span class="hljs-number">639.73</span>, <span class="hljs-number">473.76</span>] | |
| Detected cat <span class="hljs-keyword">with</span> confidence <span class="hljs-number">0.999</span> at location [<span class="hljs-number">13.24</span>, <span class="hljs-number">52.05</span>, <span class="hljs-number">314.02</span>, <span class="hljs-number">470.93</span>] | |
| Detected cat <span class="hljs-keyword">with</span> confidence <span class="hljs-number">0.999</span> at location [<span class="hljs-number">345.4</span>, <span class="hljs-number">23.85</span>, <span class="hljs-number">640.37</span>, <span class="hljs-number">368.72</span>]`,wrap:!1}}),{c(){c=a("p"),c.textContent=x,b=s(),p(y.$$.fragment)},l(l){c=i(l,"P",{"data-svelte-h":!0}),m(c)!=="svelte-kvfsh7"&&(c.textContent=x),b=r(l),h(y.$$.fragment,l)},m(l,M){d(l,c,M),d(l,b,M),g(y,l,M),D=!0},p:Lt,i(l){D||(f(y.$$.fragment,l),D=!0)},o(l){u(y.$$.fragment,l),D=!1},d(l){l&&(t(c),t(b)),_(y,l)}}}function ta(k){let c,x=`Although the recipe for forward pass needs to be defined within this function, one should call the <code>Module</code> | |
| instance afterwards instead of this since the former takes care of running the pre and post processing steps while | |
| the latter silently ignores them.`;return{c(){c=a("p"),c.innerHTML=x},l(b){c=i(b,"P",{"data-svelte-h":!0}),m(c)!=="svelte-fincs2"&&(c.innerHTML=x)},m(b,y){d(b,c,y)},p:Lt,d(b){b&&t(c)}}}function oa(k){let c,x="Examples:",b,y,D;return y=new Jt({props:{code:"aW1wb3J0JTIwaW8lMEFpbXBvcnQlMjByZXF1ZXN0cyUwQWZyb20lMjBQSUwlMjBpbXBvcnQlMjBJbWFnZSUwQWltcG9ydCUyMHRvcmNoJTBBaW1wb3J0JTIwbnVtcHklMEElMEFmcm9tJTIwdHJhbnNmb3JtZXJzJTIwaW1wb3J0JTIwQXV0b0ltYWdlUHJvY2Vzc29yJTJDJTIwRGV0ckZvclNlZ21lbnRhdGlvbiUwQWZyb20lMjB0cmFuc2Zvcm1lcnMuaW1hZ2VfdHJhbnNmb3JtcyUyMGltcG9ydCUyMHJnYl90b19pZCUwQSUwQXVybCUyMCUzRCUyMCUyMmh0dHAlM0ElMkYlMkZpbWFnZXMuY29jb2RhdGFzZXQub3JnJTJGdmFsMjAxNyUyRjAwMDAwMDAzOTc2OS5qcGclMjIlMEFpbWFnZSUyMCUzRCUyMEltYWdlLm9wZW4ocmVxdWVzdHMuZ2V0KHVybCUyQyUyMHN0cmVhbSUzRFRydWUpLnJhdyklMEElMEFpbWFnZV9wcm9jZXNzb3IlMjAlM0QlMjBBdXRvSW1hZ2VQcm9jZXNzb3IuZnJvbV9wcmV0cmFpbmVkKCUyMmZhY2Vib29rJTJGZGV0ci1yZXNuZXQtNTAtcGFub3B0aWMlMjIpJTBBbW9kZWwlMjAlM0QlMjBEZXRyRm9yU2VnbWVudGF0aW9uLmZyb21fcHJldHJhaW5lZCglMjJmYWNlYm9vayUyRmRldHItcmVzbmV0LTUwLXBhbm9wdGljJTIyKSUwQSUwQSUyMyUyMHByZXBhcmUlMjBpbWFnZSUyMGZvciUyMHRoZSUyMG1vZGVsJTBBaW5wdXRzJTIwJTNEJTIwaW1hZ2VfcHJvY2Vzc29yKGltYWdlcyUzRGltYWdlJTJDJTIwcmV0dXJuX3RlbnNvcnMlM0QlMjJwdCUyMiklMEElMEElMjMlMjBmb3J3YXJkJTIwcGFzcyUwQW91dHB1dHMlMjAlM0QlMjBtb2RlbCgqKmlucHV0cyklMEElMEElMjMlMjBVc2UlMjB0aGUlMjAlNjBwb3N0X3Byb2Nlc3NfcGFub3B0aWNfc2VnbWVudGF0aW9uJTYwJTIwbWV0aG9kJTIwb2YlMjB0aGUlMjAlNjBpbWFnZV9wcm9jZXNzb3IlNjAlMjB0byUyMHJldHJpZXZlJTIwcG9zdC1wcm9jZXNzZWQlMjBwYW5vcHRpYyUyMHNlZ21lbnRhdGlvbiUyMG1hcHMlMEElMjMlMjBTZWdtZW50YXRpb24lMjByZXN1bHRzJTIwYXJlJTIwcmV0dXJuZWQlMjBhcyUyMGElMjBsaXN0JTIwb2YlMjBkaWN0aW9uYXJpZXMlMEFyZXN1bHQlMjAlM0QlMjBpbWFnZV9wcm9jZXNzb3IucG9zdF9wcm9jZXNzX3Bhbm9wdGljX3NlZ21lbnRhdGlvbihvdXRwdXRzJTJDJTIwdGFyZ2V0X3NpemVzJTNEJTVCKDMwMCUyQyUyMDUwMCklNUQpJTBBJTBBJTIzJTIwQSUyMHRlbnNvciUyMG9mJTIwc2hhcGUlMjAoaGVpZ2h0JTJDJTIwd2lkdGgpJTIwd2hlcmUlMjBlYWNoJTIwdmFsdWUlMjBkZW5vdGVzJTIwYSUyMHNlZ21lbnQlMjBpZCUyQyUyMGZpbGxlZCUyMHdpdGglMjAtMSUyMGlmJTIwbm8lMjBzZWdtZW50JTIwaXMlMjBmb3VuZCUwQXBhbm9wdGljX3NlZyUyMCUzRCUyMHJlc3VsdCU1QjAlNUQlNUIlMjJzZWdtZW50YXRpb24lMjIlNUQlMEElMjMlMjBHZXQlMjBwcmVkaWN0aW9uJTIwc2NvcmUlMjBhbmQlMjBzZWdtZW50X2lkJTIwdG8lMjBjbGFzc19pZCUyMG1hcHBpbmclMjBvZiUyMGVhY2glMjBzZWdtZW50JTBBcGFub3B0aWNfc2VnbWVudHNfaW5mbyUyMCUzRCUyMHJlc3VsdCU1QjAlNUQlNUIlMjJzZWdtZW50c19pbmZvJTIyJTVE",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> io | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> requests | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> PIL <span class="hljs-keyword">import</span> Image | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> torch | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">import</span> numpy | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoImageProcessor, DetrForSegmentation | |
| <span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers.image_transforms <span class="hljs-keyword">import</span> rgb_to_id | |
| <span class="hljs-meta">>>> </span>url = <span class="hljs-string">"http://images.cocodataset.org/val2017/000000039769.jpg"</span> | |
| <span class="hljs-meta">>>> </span>image = Image.<span class="hljs-built_in">open</span>(requests.get(url, stream=<span class="hljs-literal">True</span>).raw) | |
| <span class="hljs-meta">>>> </span>image_processor = AutoImageProcessor.from_pretrained(<span class="hljs-string">"facebook/detr-resnet-50-panoptic"</span>) | |
| <span class="hljs-meta">>>> </span>model = DetrForSegmentation.from_pretrained(<span class="hljs-string">"facebook/detr-resnet-50-panoptic"</span>) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># prepare image for the model</span> | |
| <span class="hljs-meta">>>> </span>inputs = image_processor(images=image, return_tensors=<span class="hljs-string">"pt"</span>) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># forward pass</span> | |
| <span class="hljs-meta">>>> </span>outputs = model(**inputs) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Use the \`post_process_panoptic_segmentation\` method of the \`image_processor\` to retrieve post-processed panoptic segmentation maps</span> | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Segmentation results are returned as a list of dictionaries</span> | |
| <span class="hljs-meta">>>> </span>result = image_processor.post_process_panoptic_segmentation(outputs, target_sizes=[(<span class="hljs-number">300</span>, <span class="hljs-number">500</span>)]) | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># A tensor of shape (height, width) where each value denotes a segment id, filled with -1 if no segment is found</span> | |
| <span class="hljs-meta">>>> </span>panoptic_seg = result[<span class="hljs-number">0</span>][<span class="hljs-string">"segmentation"</span>] | |
| <span class="hljs-meta">>>> </span><span class="hljs-comment"># Get prediction score and segment_id to class_id mapping of each segment</span> | |
| <span class="hljs-meta">>>> </span>panoptic_segments_info = result[<span class="hljs-number">0</span>][<span class="hljs-string">"segments_info"</span>]`,wrap:!1}}),{c(){c=a("p"),c.textContent=x,b=s(),p(y.$$.fragment)},l(l){c=i(l,"P",{"data-svelte-h":!0}),m(c)!=="svelte-kvfsh7"&&(c.textContent=x),b=r(l),h(y.$$.fragment,l)},m(l,M){d(l,c,M),d(l,b,M),g(y,l,M),D=!0},p:Lt,i(l){D||(f(y.$$.fragment,l),D=!0)},o(l){u(y.$$.fragment,l),D=!1},d(l){l&&(t(c),t(b)),_(y,l)}}}function na(k){let c,x,b,y,D,l,M,Rs='<img alt="PyTorch" src="https://img.shields.io/badge/PyTorch-DE3412?style=flat&logo=pytorch&logoColor=white"/>',wo,Te,Do,ve,Zs=`The DETR model was proposed in <a href="https://arxiv.org/abs/2005.12872" rel="nofollow">End-to-End Object Detection with Transformers</a> by | |
| Nicolas Carion, Francisco Massa, Gabriel Synnaeve, Nicolas Usunier, Alexander Kirillov and Sergey Zagoruyko. DETR | |
| consists of a convolutional backbone followed by an encoder-decoder Transformer which can be trained end-to-end for | |
| object detection. It greatly simplifies a lot of the complexity of models like Faster-R-CNN and Mask-R-CNN, which use | |
| things like region proposals, non-maximum suppression procedure and anchor generation. Moreover, DETR can also be | |
| naturally extended to perform panoptic segmentation, by simply adding a mask head on top of the decoder outputs.`,xo,we,Ss="The abstract from the paper is the following:",Mo,De,Ws=`<em>We present a new method that views object detection as a direct set prediction problem. Our approach streamlines the | |
| detection pipeline, effectively removing the need for many hand-designed components like a non-maximum suppression | |
| procedure or anchor generation that explicitly encode our prior knowledge about the task. The main ingredients of the | |
| new framework, called DEtection TRansformer or DETR, are a set-based global loss that forces unique predictions via | |
| bipartite matching, and a transformer encoder-decoder architecture. Given a fixed small set of learned object queries, | |
| DETR reasons about the relations of the objects and the global image context to directly output the final set of | |
| predictions in parallel. The new model is conceptually simple and does not require a specialized library, unlike many | |
| other modern detectors. DETR demonstrates accuracy and run-time performance on par with the well-established and | |
| highly-optimized Faster RCNN baseline on the challenging COCO object detection dataset. Moreover, DETR can be easily | |
| generalized to produce panoptic segmentation in a unified manner. We show that it significantly outperforms competitive | |
| baselines.</em>`,Fo,xe,Hs='This model was contributed by <a href="https://huggingface.co/nielsr" rel="nofollow">nielsr</a>. The original code can be found <a href="https://github.com/facebookresearch/detr" rel="nofollow">here</a>.',jo,Me,zo,Fe,Bs='Here’s a TLDR explaining how <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForObjectDetection">DetrForObjectDetection</a> works:',$o,je,As=`First, an image is sent through a pre-trained convolutional backbone (in the paper, the authors use | |
| ResNet-50/ResNet-101). Let’s assume we also add a batch dimension. This means that the input to the backbone is a | |
| tensor of shape <code>(batch_size, 3, height, width)</code>, assuming the image has 3 color channels (RGB). The CNN backbone | |
| outputs a new lower-resolution feature map, typically of shape <code>(batch_size, 2048, height/32, width/32)</code>. This is | |
| then projected to match the hidden dimension of the Transformer of DETR, which is <code>256</code> by default, using a | |
| <code>nn.Conv2D</code> layer. So now, we have a tensor of shape <code>(batch_size, 256, height/32, width/32).</code> Next, the | |
| feature map is flattened and transposed to obtain a tensor of shape <code>(batch_size, seq_len, d_model)</code> = | |
| <code>(batch_size, width/32*height/32, 256)</code>. So a difference with NLP models is that the sequence length is actually | |
| longer than usual, but with a smaller <code>d_model</code> (which in NLP is typically 768 or higher).`,ko,ze,Gs=`Next, this is sent through the encoder, outputting <code>encoder_hidden_states</code> of the same shape (you can consider | |
| these as image features). Next, so-called <strong>object queries</strong> are sent through the decoder. This is a tensor of shape | |
| <code>(batch_size, num_queries, d_model)</code>, with <code>num_queries</code> typically set to 100 and initialized with zeros. | |
| These input embeddings are learnt positional encodings that the authors refer to as object queries, and similarly to | |
| the encoder, they are added to the input of each attention layer. Each object query will look for a particular object | |
| in the image. The decoder updates these embeddings through multiple self-attention and encoder-decoder attention layers | |
| to output <code>decoder_hidden_states</code> of the same shape: <code>(batch_size, num_queries, d_model)</code>. Next, two heads | |
| are added on top for object detection: a linear layer for classifying each object query into one of the objects or “no | |
| object”, and a MLP to predict bounding boxes for each query.`,Co,$e,Vs=`The model is trained using a <strong>bipartite matching loss</strong>: so what we actually do is compare the predicted classes + | |
| bounding boxes of each of the N = 100 object queries to the ground truth annotations, padded up to the same length N | |
| (so if an image only contains 4 objects, 96 annotations will just have a “no object” as class and “no bounding box” as | |
| bounding box). The <a href="https://en.wikipedia.org/wiki/Hungarian_algorithm" rel="nofollow">Hungarian matching algorithm</a> is used to find | |
| an optimal one-to-one mapping of each of the N queries to each of the N annotations. Next, standard cross-entropy (for | |
| the classes) and a linear combination of the L1 and <a href="https://giou.stanford.edu/" rel="nofollow">generalized IoU loss</a> (for the | |
| bounding boxes) are used to optimize the parameters of the model.`,Io,ke,Xs=`DETR can be naturally extended to perform panoptic segmentation (which unifies semantic segmentation and instance | |
| segmentation). <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a> adds a segmentation mask head on top of | |
| <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForObjectDetection">DetrForObjectDetection</a>. The mask head can be trained either jointly, or in a two steps process, | |
| where one first trains a <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForObjectDetection">DetrForObjectDetection</a> model to detect bounding boxes around both | |
| “things” (instances) and “stuff” (background things like trees, roads, sky), then freeze all the weights and train only | |
| the mask head for 25 epochs. Experimentally, these two approaches give similar results. Note that predicting boxes is | |
| required for the training to be possible, since the Hungarian matching is computed using distances between boxes.`,Po,Ce,qo,Ie,Ys=`<li>DETR uses so-called <strong>object queries</strong> to detect objects in an image. The number of queries determines the maximum | |
| number of objects that can be detected in a single image, and is set to 100 by default (see parameter | |
| <code>num_queries</code> of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrConfig">DetrConfig</a>). Note that it’s good to have some slack (in COCO, the | |
| authors used 100, while the maximum number of objects in a COCO image is ~70).</li> <li>The decoder of DETR updates the query embeddings in parallel. This is different from language models like GPT-2, | |
| which use autoregressive decoding instead of parallel. Hence, no causal attention mask is used.</li> <li>DETR adds position embeddings to the hidden states at each self-attention and cross-attention layer before projecting | |
| to queries and keys. For the position embeddings of the image, one can choose between fixed sinusoidal or learned | |
| absolute position embeddings. By default, the parameter <code>position_embedding_type</code> of | |
| <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrConfig">DetrConfig</a> is set to <code>"sine"</code>.</li> <li>During training, the authors of DETR did find it helpful to use auxiliary losses in the decoder, especially to help | |
| the model output the correct number of objects of each class. If you set the parameter <code>auxiliary_loss</code> of | |
| <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrConfig">DetrConfig</a> to <code>True</code>, then prediction feedforward neural networks and Hungarian losses | |
| are added after each decoder layer (with the FFNs sharing parameters).</li> <li>If you want to train the model in a distributed environment across multiple nodes, then one should update the | |
| <em>num_boxes</em> variable in the <em>DetrLoss</em> class of <em>modeling_detr.py</em>. When training on multiple nodes, this should be | |
| set to the average number of target boxes across all nodes, as can be seen in the original implementation <a href="https://github.com/facebookresearch/detr/blob/a54b77800eb8e64e3ad0d8237789fcbf2f8350c5/models/detr.py#L227-L232" rel="nofollow">here</a>.</li> <li><a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForObjectDetection">DetrForObjectDetection</a> and <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a> can be initialized with | |
| any convolutional backbone available in the <a href="https://github.com/rwightman/pytorch-image-models" rel="nofollow">timm library</a>. | |
| Initializing with a MobileNet backbone for example can be done by setting the <code>backbone</code> attribute of | |
| <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrConfig">DetrConfig</a> to <code>"tf_mobilenetv3_small_075"</code>, and then initializing the model with that | |
| config.</li> <li>DETR resizes the input images such that the shortest side is at least a certain amount of pixels while the longest is | |
| at most 1333 pixels. At training time, scale augmentation is used such that the shortest side is randomly set to at | |
| least 480 and at most 800 pixels. At inference time, the shortest side is set to 800. One can use | |
| <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor">DetrImageProcessor</a> to prepare images (and optional annotations in COCO format) for the | |
| model. Due to this resizing, images in a batch can have different sizes. DETR solves this by padding images up to the | |
| largest size in a batch, and by creating a pixel mask that indicates which pixels are real/which are padding. | |
| Alternatively, one can also define a custom <code>collate_fn</code> in order to batch images together, using | |
| <code>~transformers.DetrImageProcessor.pad_and_create_pixel_mask</code>.</li> <li>The size of the images will determine the amount of memory being used, and will thus determine the <code>batch_size</code>. | |
| It is advised to use a batch size of 2 per GPU. See <a href="https://github.com/facebookresearch/detr/issues/150" rel="nofollow">this Github thread</a> for more info.</li>`,Oo,Pe,Qs="There are three ways to instantiate a DETR model (depending on what you prefer):",No,qe,Ks="Option 1: Instantiate DETR with pre-trained weights for entire model",Jo,Oe,Lo,Ne,er="Option 2: Instantiate DETR with randomly initialized weights for Transformer, but pre-trained weights for backbone",Uo,Je,Eo,Le,tr="Option 3: Instantiate DETR with randomly initialized weights for backbone + Transformer",Ro,Ue,Zo,Ee,or="As a summary, consider the following table:",So,Re,nr='<thead><tr><th>Task</th> <th>Object detection</th> <th>Instance segmentation</th> <th>Panoptic segmentation</th></tr></thead> <tbody><tr><td><strong>Description</strong></td> <td>Predicting bounding boxes and class labels around objects in an image</td> <td>Predicting masks around objects (i.e. instances) in an image</td> <td>Predicting masks around both objects (i.e. instances) as well as “stuff” (i.e. background things like trees and roads) in an image</td></tr> <tr><td><strong>Model</strong></td> <td><a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForObjectDetection">DetrForObjectDetection</a></td> <td><a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a></td> <td><a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a></td></tr> <tr><td><strong>Example dataset</strong></td> <td>COCO detection</td> <td>COCO detection, COCO panoptic</td> <td>COCO panoptic</td></tr> <tr><td><strong>Format of annotations to provide to</strong> <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor">DetrImageProcessor</a></td> <td>{‘image_id’: <code>int</code>, ‘annotations’: <code>List[Dict]</code>} each Dict being a COCO object annotation</td> <td>{‘image_id’: <code>int</code>, ‘annotations’: <code>List[Dict]</code>} (in case of COCO detection) or {‘file_name’: <code>str</code>, ‘image_id’: <code>int</code>, ‘segments_info’: <code>List[Dict]</code>} (in case of COCO panoptic)</td> <td>{‘file_name’: <code>str</code>, ‘image_id’: <code>int</code>, ‘segments_info’: <code>List[Dict]</code>} and masks_path (path to directory containing PNG files of the masks)</td></tr> <tr><td><strong>Postprocessing</strong> (i.e. converting the output of the model to Pascal VOC format)</td> <td><code>post_process()</code></td> <td><code>post_process_segmentation()</code></td> <td><code>post_process_segmentation()</code>, <code>post_process_panoptic()</code></td></tr> <tr><td><strong>evaluators</strong></td> <td><code>CocoEvaluator</code> with <code>iou_types="bbox"</code></td> <td><code>CocoEvaluator</code> with <code>iou_types="bbox"</code> or <code>"segm"</code></td> <td><code>CocoEvaluator</code> with <code>iou_tupes="bbox"</code> or <code>"segm"</code>, <code>PanopticEvaluator</code></td></tr></tbody>',Wo,Ze,sr=`In short, one should prepare the data either in COCO detection or COCO panoptic format, then use | |
| <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor">DetrImageProcessor</a> to create <code>pixel_values</code>, <code>pixel_mask</code> and optional | |
| <code>labels</code>, which can then be used to train (or fine-tune) a model. For evaluation, one should first convert the | |
| outputs of the model using one of the postprocessing methods of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor">DetrImageProcessor</a>. These can | |
| be provided to either <code>CocoEvaluator</code> or <code>PanopticEvaluator</code>, which allow you to calculate metrics like | |
| mean Average Precision (mAP) and Panoptic Quality (PQ). The latter objects are implemented in the <a href="https://github.com/facebookresearch/detr" rel="nofollow">original repository</a>. See the <a href="https://github.com/NielsRogge/Transformers-Tutorials/tree/master/DETR" rel="nofollow">example notebooks</a> for more info regarding evaluation.`,Ho,Se,Bo,We,rr="A list of official Hugging Face and community (indicated by 🌎) resources to help you get started with DETR.",Ao,He,Go,Be,ar='<li>All example notebooks illustrating fine-tuning <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForObjectDetection">DetrForObjectDetection</a> and <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a> on a custom dataset can be found <a href="https://github.com/NielsRogge/Transformers-Tutorials/tree/master/DETR" rel="nofollow">here</a>.</li> <li>Scripts for finetuning <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForObjectDetection">DetrForObjectDetection</a> with <a href="/docs/transformers/pr_36839/en/main_classes/trainer#transformers.Trainer">Trainer</a> or <a href="https://huggingface.co/docs/accelerate/index" rel="nofollow">Accelerate</a> can be found <a href="https://github.com/huggingface/transformers/tree/main/examples/pytorch/object-detection" rel="nofollow">here</a>.</li> <li>See also: <a href="../tasks/object_detection">Object detection task guide</a>.</li>',Vo,Ae,ir="If you’re interested in submitting a resource to be included here, please feel free to open a Pull Request and we’ll review it! The resource should ideally demonstrate something new instead of duplicating an existing resource.",Xo,Ge,Yo,C,Ve,Pn,Ut,dr=`This is the configuration class to store the configuration of a <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrModel">DetrModel</a>. It is used to instantiate a DETR | |
| model according to the specified arguments, defining the model architecture. Instantiating a configuration with the | |
| defaults will yield a similar configuration to that of the DETR | |
| <a href="https://huggingface.co/facebook/detr-resnet-50" rel="nofollow">facebook/detr-resnet-50</a> architecture.`,qn,Et,cr=`Configuration objects inherit from <a href="/docs/transformers/pr_36839/en/main_classes/configuration#transformers.PretrainedConfig">PretrainedConfig</a> and can be used to control the model outputs. Read the | |
| documentation from <a href="/docs/transformers/pr_36839/en/main_classes/configuration#transformers.PretrainedConfig">PretrainedConfig</a> for more information.`,On,G,Nn,V,Xe,Jn,Rt,lr='Instantiate a <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrConfig">DetrConfig</a> (or a derived class) from a pre-trained backbone model configuration.',Qo,Ye,Ko,F,Qe,Ln,Zt,mr="Constructs a Detr image processor.",Un,X,Ke,En,St,pr="Preprocess an image or a batch of images so that it can be used by the model.",Rn,Y,et,Zn,Wt,hr=`Converts the raw output of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForObjectDetection">DetrForObjectDetection</a> into final bounding boxes in (top_left_x, top_left_y, | |
| bottom_right_x, bottom_right_y) format. Only supports PyTorch.`,Sn,Q,tt,Wn,Ht,gr='Converts the output of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a> into semantic segmentation maps. Only supports PyTorch.',Hn,K,ot,Bn,Bt,fr='Converts the output of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a> into instance segmentation predictions. Only supports PyTorch.',An,ee,nt,Gn,At,ur=`Converts the output of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a> into image panoptic segmentation predictions. Only supports | |
| PyTorch.`,en,st,tn,j,rt,Vn,Gt,_r="Constructs a fast Detr image processor.",Xn,te,at,Yn,Vt,br="Preprocess an image or batch of images.",Qn,oe,it,Kn,Xt,yr=`Converts the raw output of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForObjectDetection">DetrForObjectDetection</a> into final bounding boxes in (top_left_x, top_left_y, | |
| bottom_right_x, bottom_right_y) format. Only supports PyTorch.`,es,ne,dt,ts,Yt,Tr='Converts the output of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a> into semantic segmentation maps. Only supports PyTorch.',os,se,ct,ns,Qt,vr='Converts the output of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a> into instance segmentation predictions. Only supports PyTorch.',ss,re,lt,rs,Kt,wr=`Converts the output of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a> into image panoptic segmentation predictions. Only supports | |
| PyTorch.`,on,mt,nn,z,pt,as,ae,ht,is,eo,Dr="Preprocess an image or a batch of images.",ds,ie,gt,cs,to,xr=`Converts the raw output of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForObjectDetection">DetrForObjectDetection</a> into final bounding boxes in (top_left_x, top_left_y, | |
| bottom_right_x, bottom_right_y) format. Only supports PyTorch.`,ls,de,ft,ms,oo,Mr='Converts the output of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a> into semantic segmentation maps. Only supports PyTorch.',ps,ce,ut,hs,no,Fr='Converts the output of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a> into instance segmentation predictions. Only supports PyTorch.',gs,le,_t,fs,so,jr=`Converts the output of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a> into image panoptic segmentation predictions. Only supports | |
| PyTorch.`,sn,bt,rn,H,yt,us,ro,zr=`Base class for outputs of the DETR encoder-decoder model. This class adds one attribute to Seq2SeqModelOutput, | |
| namely an optional stack of intermediate decoder activations, i.e. the output of each decoder layer, each of them | |
| gone through a layernorm. This is useful when training the model with auxiliary decoding losses.`,an,B,Tt,_s,ao,$r='Output type of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForObjectDetection">DetrForObjectDetection</a>.',dn,A,vt,bs,io,kr='Output type of <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a>.',cn,wt,ln,I,Dt,ys,co,Cr=`The bare DETR Model (consisting of a backbone and encoder-decoder Transformer) outputting raw hidden-states without | |
| any specific head on top.`,Ts,lo,Ir=`This model inherits from <a href="/docs/transformers/pr_36839/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the | |
| library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads | |
| etc.)`,vs,mo,Pr=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass. | |
| Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage | |
| and behavior.`,ws,L,xt,Ds,po,qr='The <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrModel">DetrModel</a> forward method, overrides the <code>__call__</code> special method.',xs,me,Ms,pe,mn,Mt,pn,P,Ft,Fs,ho,Or=`DETR Model (consisting of a backbone and encoder-decoder Transformer) with object detection heads on top, for tasks | |
| such as COCO detection.`,js,go,Nr=`This model inherits from <a href="/docs/transformers/pr_36839/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the | |
| library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads | |
| etc.)`,zs,fo,Jr=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass. | |
| Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage | |
| and behavior.`,$s,U,jt,ks,uo,Lr='The <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForObjectDetection">DetrForObjectDetection</a> forward method, overrides the <code>__call__</code> special method.',Cs,he,Is,ge,hn,zt,gn,q,$t,Ps,_o,Ur=`DETR Model (consisting of a backbone and encoder-decoder Transformer) with a segmentation head on top, for tasks | |
| such as COCO panoptic.`,qs,bo,Er=`This model inherits from <a href="/docs/transformers/pr_36839/en/main_classes/model#transformers.PreTrainedModel">PreTrainedModel</a>. Check the superclass documentation for the generic methods the | |
| library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads | |
| etc.)`,Os,yo,Rr=`This model is also a PyTorch <a href="https://pytorch.org/docs/stable/nn.html#torch.nn.Module" rel="nofollow">torch.nn.Module</a> subclass. | |
| Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage | |
| and behavior.`,Ns,E,kt,Js,To,Zr='The <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a> forward method, overrides the <code>__call__</code> special method.',Ls,fe,Us,ue,fn,Ct,un,vo,_n;return D=new J({props:{title:"DETR",local:"detr",headingTag:"h1"}}),Te=new J({props:{title:"Overview",local:"overview",headingTag:"h2"}}),Me=new J({props:{title:"How DETR works",local:"how-detr-works",headingTag:"h2"}}),Ce=new J({props:{title:"Usage tips",local:"usage-tips",headingTag:"h2"}}),Oe=new Jt({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMERldHJGb3JPYmplY3REZXRlY3Rpb24lMEElMEFtb2RlbCUyMCUzRCUyMERldHJGb3JPYmplY3REZXRlY3Rpb24uZnJvbV9wcmV0cmFpbmVkKCUyMmZhY2Vib29rJTJGZGV0ci1yZXNuZXQtNTAlMjIp",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> DetrForObjectDetection | |
| <span class="hljs-meta">>>> </span>model = DetrForObjectDetection.from_pretrained(<span class="hljs-string">"facebook/detr-resnet-50"</span>)`,wrap:!1}}),Je=new Jt({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMERldHJDb25maWclMkMlMjBEZXRyRm9yT2JqZWN0RGV0ZWN0aW9uJTBBJTBBY29uZmlnJTIwJTNEJTIwRGV0ckNvbmZpZygpJTBBbW9kZWwlMjAlM0QlMjBEZXRyRm9yT2JqZWN0RGV0ZWN0aW9uKGNvbmZpZyk=",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> DetrConfig, DetrForObjectDetection | |
| <span class="hljs-meta">>>> </span>config = DetrConfig() | |
| <span class="hljs-meta">>>> </span>model = DetrForObjectDetection(config)`,wrap:!1}}),Ue=new Jt({props:{code:"Y29uZmlnJTIwJTNEJTIwRGV0ckNvbmZpZyh1c2VfcHJldHJhaW5lZF9iYWNrYm9uZSUzREZhbHNlKSUwQW1vZGVsJTIwJTNEJTIwRGV0ckZvck9iamVjdERldGVjdGlvbihjb25maWcp",highlighted:`<span class="hljs-meta">>>> </span>config = DetrConfig(use_pretrained_backbone=<span class="hljs-literal">False</span>) | |
| <span class="hljs-meta">>>> </span>model = DetrForObjectDetection(config)`,wrap:!1}}),Se=new J({props:{title:"Resources",local:"resources",headingTag:"h2"}}),He=new Gr({props:{pipeline:"object-detection"}}),Ge=new J({props:{title:"DetrConfig",local:"transformers.DetrConfig",headingTag:"h2"}}),Ve=new w({props:{name:"class transformers.DetrConfig",anchor:"transformers.DetrConfig",parameters:[{name:"use_timm_backbone",val:" = True"},{name:"backbone_config",val:" = None"},{name:"num_channels",val:" = 3"},{name:"num_queries",val:" = 100"},{name:"encoder_layers",val:" = 6"},{name:"encoder_ffn_dim",val:" = 2048"},{name:"encoder_attention_heads",val:" = 8"},{name:"decoder_layers",val:" = 6"},{name:"decoder_ffn_dim",val:" = 2048"},{name:"decoder_attention_heads",val:" = 8"},{name:"encoder_layerdrop",val:" = 0.0"},{name:"decoder_layerdrop",val:" = 0.0"},{name:"is_encoder_decoder",val:" = True"},{name:"activation_function",val:" = 'relu'"},{name:"d_model",val:" = 256"},{name:"dropout",val:" = 0.1"},{name:"attention_dropout",val:" = 0.0"},{name:"activation_dropout",val:" = 0.0"},{name:"init_std",val:" = 0.02"},{name:"init_xavier_std",val:" = 1.0"},{name:"auxiliary_loss",val:" = False"},{name:"position_embedding_type",val:" = 'sine'"},{name:"backbone",val:" = 'resnet50'"},{name:"use_pretrained_backbone",val:" = True"},{name:"backbone_kwargs",val:" = None"},{name:"dilation",val:" = False"},{name:"class_cost",val:" = 1"},{name:"bbox_cost",val:" = 5"},{name:"giou_cost",val:" = 2"},{name:"mask_loss_coefficient",val:" = 1"},{name:"dice_loss_coefficient",val:" = 1"},{name:"bbox_loss_coefficient",val:" = 5"},{name:"giou_loss_coefficient",val:" = 2"},{name:"eos_coefficient",val:" = 0.1"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.DetrConfig.use_timm_backbone",description:`<strong>use_timm_backbone</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to use the <code>timm</code> library for the backbone. If set to <code>False</code>, will use the <a href="/docs/transformers/pr_36839/en/main_classes/backbones#transformers.AutoBackbone">AutoBackbone</a> | |
| API.`,name:"use_timm_backbone"},{anchor:"transformers.DetrConfig.backbone_config",description:`<strong>backbone_config</strong> (<code>PretrainedConfig</code> or <code>dict</code>, <em>optional</em>) — | |
| The configuration of the backbone model. Only used in case <code>use_timm_backbone</code> is set to <code>False</code> in which | |
| case it will default to <code>ResNetConfig()</code>.`,name:"backbone_config"},{anchor:"transformers.DetrConfig.num_channels",description:`<strong>num_channels</strong> (<code>int</code>, <em>optional</em>, defaults to 3) — | |
| The number of input channels.`,name:"num_channels"},{anchor:"transformers.DetrConfig.num_queries",description:`<strong>num_queries</strong> (<code>int</code>, <em>optional</em>, defaults to 100) — | |
| Number of object queries, i.e. detection slots. This is the maximal number of objects <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrModel">DetrModel</a> can | |
| detect in a single image. For COCO, we recommend 100 queries.`,name:"num_queries"},{anchor:"transformers.DetrConfig.d_model",description:`<strong>d_model</strong> (<code>int</code>, <em>optional</em>, defaults to 256) — | |
| This parameter is a general dimension parameter, defining dimensions for components such as the encoder layer and projection parameters in the decoder layer, among others.`,name:"d_model"},{anchor:"transformers.DetrConfig.encoder_layers",description:`<strong>encoder_layers</strong> (<code>int</code>, <em>optional</em>, defaults to 6) — | |
| Number of encoder layers.`,name:"encoder_layers"},{anchor:"transformers.DetrConfig.decoder_layers",description:`<strong>decoder_layers</strong> (<code>int</code>, <em>optional</em>, defaults to 6) — | |
| Number of decoder layers.`,name:"decoder_layers"},{anchor:"transformers.DetrConfig.encoder_attention_heads",description:`<strong>encoder_attention_heads</strong> (<code>int</code>, <em>optional</em>, defaults to 8) — | |
| Number of attention heads for each attention layer in the Transformer encoder.`,name:"encoder_attention_heads"},{anchor:"transformers.DetrConfig.decoder_attention_heads",description:`<strong>decoder_attention_heads</strong> (<code>int</code>, <em>optional</em>, defaults to 8) — | |
| Number of attention heads for each attention layer in the Transformer decoder.`,name:"decoder_attention_heads"},{anchor:"transformers.DetrConfig.decoder_ffn_dim",description:`<strong>decoder_ffn_dim</strong> (<code>int</code>, <em>optional</em>, defaults to 2048) — | |
| Dimension of the “intermediate” (often named feed-forward) layer in decoder.`,name:"decoder_ffn_dim"},{anchor:"transformers.DetrConfig.encoder_ffn_dim",description:`<strong>encoder_ffn_dim</strong> (<code>int</code>, <em>optional</em>, defaults to 2048) — | |
| Dimension of the “intermediate” (often named feed-forward) layer in decoder.`,name:"encoder_ffn_dim"},{anchor:"transformers.DetrConfig.activation_function",description:`<strong>activation_function</strong> (<code>str</code> or <code>function</code>, <em>optional</em>, defaults to <code>"relu"</code>) — | |
| The non-linear activation function (function or string) in the encoder and pooler. If string, <code>"gelu"</code>, | |
| <code>"relu"</code>, <code>"silu"</code> and <code>"gelu_new"</code> are supported.`,name:"activation_function"},{anchor:"transformers.DetrConfig.dropout",description:`<strong>dropout</strong> (<code>float</code>, <em>optional</em>, defaults to 0.1) — | |
| The dropout probability for all fully connected layers in the embeddings, encoder, and pooler.`,name:"dropout"},{anchor:"transformers.DetrConfig.attention_dropout",description:`<strong>attention_dropout</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| The dropout ratio for the attention probabilities.`,name:"attention_dropout"},{anchor:"transformers.DetrConfig.activation_dropout",description:`<strong>activation_dropout</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| The dropout ratio for activations inside the fully connected layer.`,name:"activation_dropout"},{anchor:"transformers.DetrConfig.init_std",description:`<strong>init_std</strong> (<code>float</code>, <em>optional</em>, defaults to 0.02) — | |
| The standard deviation of the truncated_normal_initializer for initializing all weight matrices.`,name:"init_std"},{anchor:"transformers.DetrConfig.init_xavier_std",description:`<strong>init_xavier_std</strong> (<code>float</code>, <em>optional</em>, defaults to 1) — | |
| The scaling factor used for the Xavier initialization gain in the HM Attention map module.`,name:"init_xavier_std"},{anchor:"transformers.DetrConfig.encoder_layerdrop",description:`<strong>encoder_layerdrop</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| The LayerDrop probability for the encoder. See the [LayerDrop paper](see <a href="https://arxiv.org/abs/1909.11556" rel="nofollow">https://arxiv.org/abs/1909.11556</a>) | |
| for more details.`,name:"encoder_layerdrop"},{anchor:"transformers.DetrConfig.decoder_layerdrop",description:`<strong>decoder_layerdrop</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| The LayerDrop probability for the decoder. See the [LayerDrop paper](see <a href="https://arxiv.org/abs/1909.11556" rel="nofollow">https://arxiv.org/abs/1909.11556</a>) | |
| for more details.`,name:"decoder_layerdrop"},{anchor:"transformers.DetrConfig.auxiliary_loss",description:`<strong>auxiliary_loss</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether auxiliary decoding losses (loss at each decoder layer) are to be used.`,name:"auxiliary_loss"},{anchor:"transformers.DetrConfig.position_embedding_type",description:`<strong>position_embedding_type</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"sine"</code>) — | |
| Type of position embeddings to be used on top of the image features. One of <code>"sine"</code> or <code>"learned"</code>.`,name:"position_embedding_type"},{anchor:"transformers.DetrConfig.backbone",description:`<strong>backbone</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"resnet50"</code>) — | |
| Name of backbone to use when <code>backbone_config</code> is <code>None</code>. If <code>use_pretrained_backbone</code> is <code>True</code>, this | |
| will load the corresponding pretrained weights from the timm or transformers library. If <code>use_pretrained_backbone</code> | |
| is <code>False</code>, this loads the backbone’s config and uses that to initialize the backbone with random weights.`,name:"backbone"},{anchor:"transformers.DetrConfig.use_pretrained_backbone",description:`<strong>use_pretrained_backbone</strong> (<code>bool</code>, <em>optional</em>, <code>True</code>) — | |
| Whether to use pretrained weights for the backbone.`,name:"use_pretrained_backbone"},{anchor:"transformers.DetrConfig.backbone_kwargs",description:`<strong>backbone_kwargs</strong> (<code>dict</code>, <em>optional</em>) — | |
| Keyword arguments to be passed to AutoBackbone when loading from a checkpoint | |
| e.g. <code>{'out_indices': (0, 1, 2, 3)}</code>. Cannot be specified if <code>backbone_config</code> is set.`,name:"backbone_kwargs"},{anchor:"transformers.DetrConfig.dilation",description:`<strong>dilation</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to replace stride with dilation in the last convolutional block (DC5). Only supported when | |
| <code>use_timm_backbone</code> = <code>True</code>.`,name:"dilation"},{anchor:"transformers.DetrConfig.class_cost",description:`<strong>class_cost</strong> (<code>float</code>, <em>optional</em>, defaults to 1) — | |
| Relative weight of the classification error in the Hungarian matching cost.`,name:"class_cost"},{anchor:"transformers.DetrConfig.bbox_cost",description:`<strong>bbox_cost</strong> (<code>float</code>, <em>optional</em>, defaults to 5) — | |
| Relative weight of the L1 error of the bounding box coordinates in the Hungarian matching cost.`,name:"bbox_cost"},{anchor:"transformers.DetrConfig.giou_cost",description:`<strong>giou_cost</strong> (<code>float</code>, <em>optional</em>, defaults to 2) — | |
| Relative weight of the generalized IoU loss of the bounding box in the Hungarian matching cost.`,name:"giou_cost"},{anchor:"transformers.DetrConfig.mask_loss_coefficient",description:`<strong>mask_loss_coefficient</strong> (<code>float</code>, <em>optional</em>, defaults to 1) — | |
| Relative weight of the Focal loss in the panoptic segmentation loss.`,name:"mask_loss_coefficient"},{anchor:"transformers.DetrConfig.dice_loss_coefficient",description:`<strong>dice_loss_coefficient</strong> (<code>float</code>, <em>optional</em>, defaults to 1) — | |
| Relative weight of the DICE/F-1 loss in the panoptic segmentation loss.`,name:"dice_loss_coefficient"},{anchor:"transformers.DetrConfig.bbox_loss_coefficient",description:`<strong>bbox_loss_coefficient</strong> (<code>float</code>, <em>optional</em>, defaults to 5) — | |
| Relative weight of the L1 bounding box loss in the object detection loss.`,name:"bbox_loss_coefficient"},{anchor:"transformers.DetrConfig.giou_loss_coefficient",description:`<strong>giou_loss_coefficient</strong> (<code>float</code>, <em>optional</em>, defaults to 2) — | |
| Relative weight of the generalized IoU loss in the object detection loss.`,name:"giou_loss_coefficient"},{anchor:"transformers.DetrConfig.eos_coefficient",description:`<strong>eos_coefficient</strong> (<code>float</code>, <em>optional</em>, defaults to 0.1) — | |
| Relative classification weight of the ‘no-object’ class in the object detection loss.`,name:"eos_coefficient"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/configuration_detr.py#L32"}}),G=new In({props:{anchor:"transformers.DetrConfig.example",$$slots:{default:[Xr]},$$scope:{ctx:k}}}),Xe=new w({props:{name:"from_backbone_config",anchor:"transformers.DetrConfig.from_backbone_config",parameters:[{name:"backbone_config",val:": PretrainedConfig"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.DetrConfig.from_backbone_config.backbone_config",description:`<strong>backbone_config</strong> (<a href="/docs/transformers/pr_36839/en/main_classes/configuration#transformers.PretrainedConfig">PretrainedConfig</a>) — | |
| The backbone configuration.`,name:"backbone_config"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/configuration_detr.py#L255",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>An instance of a configuration object</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrConfig" | |
| >DetrConfig</a></p> | |
| `}}),Ye=new J({props:{title:"DetrImageProcessor",local:"transformers.DetrImageProcessor",headingTag:"h2"}}),Qe=new w({props:{name:"class transformers.DetrImageProcessor",anchor:"transformers.DetrImageProcessor",parameters:[{name:"format",val:": typing.Union[str, transformers.image_utils.AnnotationFormat] = <AnnotationFormat.COCO_DETECTION: 'coco_detection'>"},{name:"do_resize",val:": bool = True"},{name:"size",val:": typing.Dict[str, int] = None"},{name:"resample",val:": Resampling = <Resampling.BILINEAR: 2>"},{name:"do_rescale",val:": bool = True"},{name:"rescale_factor",val:": typing.Union[int, float] = 0.00392156862745098"},{name:"do_normalize",val:": bool = True"},{name:"image_mean",val:": typing.Union[float, typing.List[float]] = None"},{name:"image_std",val:": typing.Union[float, typing.List[float]] = None"},{name:"do_convert_annotations",val:": typing.Optional[bool] = None"},{name:"do_pad",val:": bool = True"},{name:"pad_size",val:": typing.Optional[typing.Dict[str, int]] = None"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.DetrImageProcessor.format",description:`<strong>format</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"coco_detection"</code>) — | |
| Data format of the annotations. One of “coco_detection” or “coco_panoptic”.`,name:"format"},{anchor:"transformers.DetrImageProcessor.do_resize",description:`<strong>do_resize</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Controls whether to resize the image’s <code>(height, width)</code> dimensions to the specified <code>size</code>. Can be | |
| overridden by the <code>do_resize</code> parameter in the <code>preprocess</code> method.`,name:"do_resize"},{anchor:"transformers.DetrImageProcessor.size",description:`<strong>size</strong> (<code>Dict[str, int]</code> <em>optional</em>, defaults to <code>{"shortest_edge" -- 800, "longest_edge": 1333}</code>): | |
| Size of the image’s <code>(height, width)</code> dimensions after resizing. Can be overridden by the <code>size</code> parameter | |
| in the <code>preprocess</code> method. Available options are:<ul> | |
| <li><code>{"height": int, "width": int}</code>: The image will be resized to the exact size <code>(height, width)</code>. | |
| Do NOT keep the aspect ratio.</li> | |
| <li><code>{"shortest_edge": int, "longest_edge": int}</code>: The image will be resized to a maximum size respecting | |
| the aspect ratio and keeping the shortest edge less or equal to <code>shortest_edge</code> and the longest edge | |
| less or equal to <code>longest_edge</code>.</li> | |
| <li><code>{"max_height": int, "max_width": int}</code>: The image will be resized to the maximum size respecting the | |
| aspect ratio and keeping the height less or equal to <code>max_height</code> and the width less or equal to | |
| <code>max_width</code>.</li> | |
| </ul>`,name:"size"},{anchor:"transformers.DetrImageProcessor.resample",description:`<strong>resample</strong> (<code>PILImageResampling</code>, <em>optional</em>, defaults to <code>PILImageResampling.BILINEAR</code>) — | |
| Resampling filter to use if resizing the image.`,name:"resample"},{anchor:"transformers.DetrImageProcessor.do_rescale",description:`<strong>do_rescale</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Controls whether to rescale the image by the specified scale <code>rescale_factor</code>. Can be overridden by the | |
| <code>do_rescale</code> parameter in the <code>preprocess</code> method.`,name:"do_rescale"},{anchor:"transformers.DetrImageProcessor.rescale_factor",description:`<strong>rescale_factor</strong> (<code>int</code> or <code>float</code>, <em>optional</em>, defaults to <code>1/255</code>) — | |
| Scale factor to use if rescaling the image. Can be overridden by the <code>rescale_factor</code> parameter in the | |
| <code>preprocess</code> method.`,name:"rescale_factor"},{anchor:"transformers.DetrImageProcessor.do_normalize",description:`<strong>do_normalize</strong> (<code>bool</code>, <em>optional</em>, defaults to True) — | |
| Controls whether to normalize the image. Can be overridden by the <code>do_normalize</code> parameter in the | |
| <code>preprocess</code> method.`,name:"do_normalize"},{anchor:"transformers.DetrImageProcessor.image_mean",description:`<strong>image_mean</strong> (<code>float</code> or <code>List[float]</code>, <em>optional</em>, defaults to <code>IMAGENET_DEFAULT_MEAN</code>) — | |
| Mean values to use when normalizing the image. Can be a single value or a list of values, one for each | |
| channel. Can be overridden by the <code>image_mean</code> parameter in the <code>preprocess</code> method.`,name:"image_mean"},{anchor:"transformers.DetrImageProcessor.image_std",description:`<strong>image_std</strong> (<code>float</code> or <code>List[float]</code>, <em>optional</em>, defaults to <code>IMAGENET_DEFAULT_STD</code>) — | |
| Standard deviation values to use when normalizing the image. Can be a single value or a list of values, one | |
| for each channel. Can be overridden by the <code>image_std</code> parameter in the <code>preprocess</code> method.`,name:"image_std"},{anchor:"transformers.DetrImageProcessor.do_convert_annotations",description:`<strong>do_convert_annotations</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Controls whether to convert the annotations to the format expected by the DETR model. Converts the | |
| bounding boxes to the format <code>(center_x, center_y, width, height)</code> and in the range <code>[0, 1]</code>. | |
| Can be overridden by the <code>do_convert_annotations</code> parameter in the <code>preprocess</code> method.`,name:"do_convert_annotations"},{anchor:"transformers.DetrImageProcessor.do_pad",description:`<strong>do_pad</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Controls whether to pad the image. Can be overridden by the <code>do_pad</code> parameter in the <code>preprocess</code> | |
| method. If <code>True</code>, padding will be applied to the bottom and right of the image with zeros. | |
| If <code>pad_size</code> is provided, the image will be padded to the specified dimensions. | |
| Otherwise, the image will be padded to the maximum height and width of the batch.`,name:"do_pad"},{anchor:"transformers.DetrImageProcessor.pad_size",description:`<strong>pad_size</strong> (<code>Dict[str, int]</code>, <em>optional</em>) — | |
| The size <code>{"height": int, "width" int}</code> to pad the images to. Must be larger than any image size | |
| provided for preprocessing. If <code>pad_size</code> is not provided, images will be padded to the largest | |
| height and width in the batch.`,name:"pad_size"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr.py#L787"}}),Ke=new w({props:{name:"preprocess",anchor:"transformers.DetrImageProcessor.preprocess",parameters:[{name:"images",val:": typing.Union[ForwardRef('PIL.Image.Image'), numpy.ndarray, ForwardRef('torch.Tensor'), typing.List[ForwardRef('PIL.Image.Image')], typing.List[numpy.ndarray], typing.List[ForwardRef('torch.Tensor')]]"},{name:"annotations",val:": typing.Union[typing.Dict[str, typing.Union[int, str, typing.List[typing.Dict]]], typing.List[typing.Dict[str, typing.Union[int, str, typing.List[typing.Dict]]]], NoneType] = None"},{name:"return_segmentation_masks",val:": bool = None"},{name:"masks_path",val:": typing.Union[str, pathlib.Path, NoneType] = None"},{name:"do_resize",val:": typing.Optional[bool] = None"},{name:"size",val:": typing.Optional[typing.Dict[str, int]] = None"},{name:"resample",val:" = None"},{name:"do_rescale",val:": typing.Optional[bool] = None"},{name:"rescale_factor",val:": typing.Union[int, float, NoneType] = None"},{name:"do_normalize",val:": typing.Optional[bool] = None"},{name:"do_convert_annotations",val:": typing.Optional[bool] = None"},{name:"image_mean",val:": typing.Union[float, typing.List[float], NoneType] = None"},{name:"image_std",val:": typing.Union[float, typing.List[float], NoneType] = None"},{name:"do_pad",val:": typing.Optional[bool] = None"},{name:"format",val:": typing.Union[str, transformers.image_utils.AnnotationFormat, NoneType] = None"},{name:"return_tensors",val:": typing.Union[str, transformers.utils.generic.TensorType, NoneType] = None"},{name:"data_format",val:": typing.Union[str, transformers.image_utils.ChannelDimension] = <ChannelDimension.FIRST: 'channels_first'>"},{name:"input_data_format",val:": typing.Union[str, transformers.image_utils.ChannelDimension, NoneType] = None"},{name:"pad_size",val:": typing.Optional[typing.Dict[str, int]] = None"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.DetrImageProcessor.preprocess.images",description:`<strong>images</strong> (<code>ImageInput</code>) — | |
| Image or batch of images to preprocess. Expects a single or batch of images with pixel values ranging | |
| from 0 to 255. If passing in images with pixel values between 0 and 1, set <code>do_rescale=False</code>.`,name:"images"},{anchor:"transformers.DetrImageProcessor.preprocess.annotations",description:`<strong>annotations</strong> (<code>AnnotationType</code> or <code>List[AnnotationType]</code>, <em>optional</em>) — | |
| List of annotations associated with the image or batch of images. If annotation is for object | |
| detection, the annotations should be a dictionary with the following keys:<ul> | |
| <li>“image_id” (<code>int</code>): The image id.</li> | |
| <li>“annotations” (<code>List[Dict]</code>): List of annotations for an image. Each annotation should be a | |
| dictionary. An image can have no annotations, in which case the list should be empty. | |
| If annotation is for segmentation, the annotations should be a dictionary with the following keys:</li> | |
| <li>“image_id” (<code>int</code>): The image id.</li> | |
| <li>“segments_info” (<code>List[Dict]</code>): List of segments for an image. Each segment should be a dictionary. | |
| An image can have no segments, in which case the list should be empty.</li> | |
| <li>“file_name” (<code>str</code>): The file name of the image.</li> | |
| </ul>`,name:"annotations"},{anchor:"transformers.DetrImageProcessor.preprocess.return_segmentation_masks",description:`<strong>return_segmentation_masks</strong> (<code>bool</code>, <em>optional</em>, defaults to self.return_segmentation_masks) — | |
| Whether to return segmentation masks.`,name:"return_segmentation_masks"},{anchor:"transformers.DetrImageProcessor.preprocess.masks_path",description:`<strong>masks_path</strong> (<code>str</code> or <code>pathlib.Path</code>, <em>optional</em>) — | |
| Path to the directory containing the segmentation masks.`,name:"masks_path"},{anchor:"transformers.DetrImageProcessor.preprocess.do_resize",description:`<strong>do_resize</strong> (<code>bool</code>, <em>optional</em>, defaults to self.do_resize) — | |
| Whether to resize the image.`,name:"do_resize"},{anchor:"transformers.DetrImageProcessor.preprocess.size",description:`<strong>size</strong> (<code>Dict[str, int]</code>, <em>optional</em>, defaults to self.size) — | |
| Size of the image’s <code>(height, width)</code> dimensions after resizing. Available options are:<ul> | |
| <li><code>{"height": int, "width": int}</code>: The image will be resized to the exact size <code>(height, width)</code>. | |
| Do NOT keep the aspect ratio.</li> | |
| <li><code>{"shortest_edge": int, "longest_edge": int}</code>: The image will be resized to a maximum size respecting | |
| the aspect ratio and keeping the shortest edge less or equal to <code>shortest_edge</code> and the longest edge | |
| less or equal to <code>longest_edge</code>.</li> | |
| <li><code>{"max_height": int, "max_width": int}</code>: The image will be resized to the maximum size respecting the | |
| aspect ratio and keeping the height less or equal to <code>max_height</code> and the width less or equal to | |
| <code>max_width</code>.</li> | |
| </ul>`,name:"size"},{anchor:"transformers.DetrImageProcessor.preprocess.resample",description:`<strong>resample</strong> (<code>PILImageResampling</code>, <em>optional</em>, defaults to self.resample) — | |
| Resampling filter to use when resizing the image.`,name:"resample"},{anchor:"transformers.DetrImageProcessor.preprocess.do_rescale",description:`<strong>do_rescale</strong> (<code>bool</code>, <em>optional</em>, defaults to self.do_rescale) — | |
| Whether to rescale the image.`,name:"do_rescale"},{anchor:"transformers.DetrImageProcessor.preprocess.rescale_factor",description:`<strong>rescale_factor</strong> (<code>float</code>, <em>optional</em>, defaults to self.rescale_factor) — | |
| Rescale factor to use when rescaling the image.`,name:"rescale_factor"},{anchor:"transformers.DetrImageProcessor.preprocess.do_normalize",description:`<strong>do_normalize</strong> (<code>bool</code>, <em>optional</em>, defaults to self.do_normalize) — | |
| Whether to normalize the image.`,name:"do_normalize"},{anchor:"transformers.DetrImageProcessor.preprocess.do_convert_annotations",description:`<strong>do_convert_annotations</strong> (<code>bool</code>, <em>optional</em>, defaults to self.do_convert_annotations) — | |
| Whether to convert the annotations to the format expected by the model. Converts the bounding | |
| boxes from the format <code>(top_left_x, top_left_y, width, height)</code> to <code>(center_x, center_y, width, height)</code> | |
| and in relative coordinates.`,name:"do_convert_annotations"},{anchor:"transformers.DetrImageProcessor.preprocess.image_mean",description:`<strong>image_mean</strong> (<code>float</code> or <code>List[float]</code>, <em>optional</em>, defaults to self.image_mean) — | |
| Mean to use when normalizing the image.`,name:"image_mean"},{anchor:"transformers.DetrImageProcessor.preprocess.image_std",description:`<strong>image_std</strong> (<code>float</code> or <code>List[float]</code>, <em>optional</em>, defaults to self.image_std) — | |
| Standard deviation to use when normalizing the image.`,name:"image_std"},{anchor:"transformers.DetrImageProcessor.preprocess.do_pad",description:`<strong>do_pad</strong> (<code>bool</code>, <em>optional</em>, defaults to self.do_pad) — | |
| Whether to pad the image. If <code>True</code>, padding will be applied to the bottom and right of | |
| the image with zeros. If <code>pad_size</code> is provided, the image will be padded to the specified | |
| dimensions. Otherwise, the image will be padded to the maximum height and width of the batch.`,name:"do_pad"},{anchor:"transformers.DetrImageProcessor.preprocess.format",description:`<strong>format</strong> (<code>str</code> or <code>AnnotationFormat</code>, <em>optional</em>, defaults to self.format) — | |
| Format of the annotations.`,name:"format"},{anchor:"transformers.DetrImageProcessor.preprocess.return_tensors",description:`<strong>return_tensors</strong> (<code>str</code> or <code>TensorType</code>, <em>optional</em>, defaults to self.return_tensors) — | |
| Type of tensors to return. If <code>None</code>, will return the list of images.`,name:"return_tensors"},{anchor:"transformers.DetrImageProcessor.preprocess.data_format",description:`<strong>data_format</strong> (<code>ChannelDimension</code> or <code>str</code>, <em>optional</em>, defaults to <code>ChannelDimension.FIRST</code>) — | |
| The channel dimension format for the output image. Can be one of:<ul> | |
| <li><code>"channels_first"</code> or <code>ChannelDimension.FIRST</code>: image in (num_channels, height, width) format.</li> | |
| <li><code>"channels_last"</code> or <code>ChannelDimension.LAST</code>: image in (height, width, num_channels) format.</li> | |
| <li>Unset: Use the channel dimension format of the input image.</li> | |
| </ul>`,name:"data_format"},{anchor:"transformers.DetrImageProcessor.preprocess.input_data_format",description:`<strong>input_data_format</strong> (<code>ChannelDimension</code> or <code>str</code>, <em>optional</em>) — | |
| The channel dimension format for the input image. If unset, the channel dimension format is inferred | |
| from the input image. Can be one of:<ul> | |
| <li><code>"channels_first"</code> or <code>ChannelDimension.FIRST</code>: image in (num_channels, height, width) format.</li> | |
| <li><code>"channels_last"</code> or <code>ChannelDimension.LAST</code>: image in (height, width, num_channels) format.</li> | |
| <li><code>"none"</code> or <code>ChannelDimension.NONE</code>: image in (height, width) format.</li> | |
| </ul>`,name:"input_data_format"},{anchor:"transformers.DetrImageProcessor.preprocess.pad_size",description:`<strong>pad_size</strong> (<code>Dict[str, int]</code>, <em>optional</em>) — | |
| The size <code>{"height": int, "width" int}</code> to pad the images to. Must be larger than any image size | |
| provided for preprocessing. If <code>pad_size</code> is not provided, images will be padded to the largest | |
| height and width in the batch.`,name:"pad_size"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr.py#L1236"}}),et=new w({props:{name:"post_process_object_detection",anchor:"transformers.DetrImageProcessor.post_process_object_detection",parameters:[{name:"outputs",val:""},{name:"threshold",val:": float = 0.5"},{name:"target_sizes",val:": typing.Union[transformers.utils.generic.TensorType, typing.List[typing.Tuple]] = None"}],parametersDescription:[{anchor:"transformers.DetrImageProcessor.post_process_object_detection.outputs",description:`<strong>outputs</strong> (<code>DetrObjectDetectionOutput</code>) — | |
| Raw outputs of the model.`,name:"outputs"},{anchor:"transformers.DetrImageProcessor.post_process_object_detection.threshold",description:`<strong>threshold</strong> (<code>float</code>, <em>optional</em>) — | |
| Score threshold to keep object detection predictions.`,name:"threshold"},{anchor:"transformers.DetrImageProcessor.post_process_object_detection.target_sizes",description:`<strong>target_sizes</strong> (<code>torch.Tensor</code> or <code>List[Tuple[int, int]]</code>, <em>optional</em>) — | |
| Tensor of shape <code>(batch_size, 2)</code> or list of tuples (<code>Tuple[int, int]</code>) containing the target size | |
| <code>(height, width)</code> of each image in the batch. If unset, predictions will not be resized.`,name:"target_sizes"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr.py#L1772",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of dictionaries, each dictionary containing the scores, labels and boxes for an image | |
| in the batch as predicted by the model.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[Dict]</code></p> | |
| `}}),tt=new w({props:{name:"post_process_semantic_segmentation",anchor:"transformers.DetrImageProcessor.post_process_semantic_segmentation",parameters:[{name:"outputs",val:""},{name:"target_sizes",val:": typing.List[typing.Tuple[int, int]] = None"}],parametersDescription:[{anchor:"transformers.DetrImageProcessor.post_process_semantic_segmentation.outputs",description:`<strong>outputs</strong> (<a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a>) — | |
| Raw outputs of the model.`,name:"outputs"},{anchor:"transformers.DetrImageProcessor.post_process_semantic_segmentation.target_sizes",description:`<strong>target_sizes</strong> (<code>List[Tuple[int, int]]</code>, <em>optional</em>) — | |
| A list of tuples (<code>Tuple[int, int]</code>) containing the target size (height, width) of each image in the | |
| batch. If unset, predictions will not be resized.`,name:"target_sizes"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr.py#L1825",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of length <code>batch_size</code>, where each item is a semantic segmentation map of shape (height, width) | |
| corresponding to the target_sizes entry (if <code>target_sizes</code> is specified). Each entry of each | |
| <code>torch.Tensor</code> correspond to a semantic class id.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[torch.Tensor]</code></p> | |
| `}}),ot=new w({props:{name:"post_process_instance_segmentation",anchor:"transformers.DetrImageProcessor.post_process_instance_segmentation",parameters:[{name:"outputs",val:""},{name:"threshold",val:": float = 0.5"},{name:"mask_threshold",val:": float = 0.5"},{name:"overlap_mask_area_threshold",val:": float = 0.8"},{name:"target_sizes",val:": typing.Optional[typing.List[typing.Tuple[int, int]]] = None"},{name:"return_coco_annotation",val:": typing.Optional[bool] = False"}],parametersDescription:[{anchor:"transformers.DetrImageProcessor.post_process_instance_segmentation.outputs",description:`<strong>outputs</strong> (<a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a>) — | |
| Raw outputs of the model.`,name:"outputs"},{anchor:"transformers.DetrImageProcessor.post_process_instance_segmentation.threshold",description:`<strong>threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.5) — | |
| The probability score threshold to keep predicted instance masks.`,name:"threshold"},{anchor:"transformers.DetrImageProcessor.post_process_instance_segmentation.mask_threshold",description:`<strong>mask_threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.5) — | |
| Threshold to use when turning the predicted masks into binary values.`,name:"mask_threshold"},{anchor:"transformers.DetrImageProcessor.post_process_instance_segmentation.overlap_mask_area_threshold",description:`<strong>overlap_mask_area_threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.8) — | |
| The overlap mask area threshold to merge or discard small disconnected parts within each binary | |
| instance mask.`,name:"overlap_mask_area_threshold"},{anchor:"transformers.DetrImageProcessor.post_process_instance_segmentation.target_sizes",description:`<strong>target_sizes</strong> (<code>List[Tuple]</code>, <em>optional</em>) — | |
| List of length (batch_size), where each list item (<code>Tuple[int, int]]</code>) corresponds to the requested | |
| final size (height, width) of each prediction. If unset, predictions will not be resized.`,name:"target_sizes"},{anchor:"transformers.DetrImageProcessor.post_process_instance_segmentation.return_coco_annotation",description:`<strong>return_coco_annotation</strong> (<code>bool</code>, <em>optional</em>) — | |
| Defaults to <code>False</code>. If set to <code>True</code>, segmentation maps are returned in COCO run-length encoding (RLE) | |
| format.`,name:"return_coco_annotation"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr.py#L1873",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of dictionaries, one per image, each dictionary containing two keys:</p> | |
| <ul> | |
| <li><strong>segmentation</strong> — A tensor of shape <code>(height, width)</code> where each pixel represents a <code>segment_id</code> or | |
| <code>List[List]</code> run-length encoding (RLE) of the segmentation map if return_coco_annotation is set to | |
| <code>True</code>. Set to <code>None</code> if no mask if found above <code>threshold</code>.</li> | |
| <li><strong>segments_info</strong> — A dictionary that contains additional information on each segment.<ul> | |
| <li><strong>id</strong> — An integer representing the <code>segment_id</code>.</li> | |
| <li><strong>label_id</strong> — An integer representing the label / semantic class id corresponding to <code>segment_id</code>.</li> | |
| <li><strong>score</strong> — Prediction score of segment with <code>segment_id</code>.</li> | |
| </ul></li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[Dict]</code></p> | |
| `}}),nt=new w({props:{name:"post_process_panoptic_segmentation",anchor:"transformers.DetrImageProcessor.post_process_panoptic_segmentation",parameters:[{name:"outputs",val:""},{name:"threshold",val:": float = 0.5"},{name:"mask_threshold",val:": float = 0.5"},{name:"overlap_mask_area_threshold",val:": float = 0.8"},{name:"label_ids_to_fuse",val:": typing.Optional[typing.Set[int]] = None"},{name:"target_sizes",val:": typing.Optional[typing.List[typing.Tuple[int, int]]] = None"}],parametersDescription:[{anchor:"transformers.DetrImageProcessor.post_process_panoptic_segmentation.outputs",description:`<strong>outputs</strong> (<a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a>) — | |
| The outputs from <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a>.`,name:"outputs"},{anchor:"transformers.DetrImageProcessor.post_process_panoptic_segmentation.threshold",description:`<strong>threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.5) — | |
| The probability score threshold to keep predicted instance masks.`,name:"threshold"},{anchor:"transformers.DetrImageProcessor.post_process_panoptic_segmentation.mask_threshold",description:`<strong>mask_threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.5) — | |
| Threshold to use when turning the predicted masks into binary values.`,name:"mask_threshold"},{anchor:"transformers.DetrImageProcessor.post_process_panoptic_segmentation.overlap_mask_area_threshold",description:`<strong>overlap_mask_area_threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.8) — | |
| The overlap mask area threshold to merge or discard small disconnected parts within each binary | |
| instance mask.`,name:"overlap_mask_area_threshold"},{anchor:"transformers.DetrImageProcessor.post_process_panoptic_segmentation.label_ids_to_fuse",description:`<strong>label_ids_to_fuse</strong> (<code>Set[int]</code>, <em>optional</em>) — | |
| The labels in this state will have all their instances be fused together. For instance we could say | |
| there can only be one sky in an image, but several persons, so the label ID for sky would be in that | |
| set, but not the one for person.`,name:"label_ids_to_fuse"},{anchor:"transformers.DetrImageProcessor.post_process_panoptic_segmentation.target_sizes",description:`<strong>target_sizes</strong> (<code>List[Tuple]</code>, <em>optional</em>) — | |
| List of length (batch_size), where each list item (<code>Tuple[int, int]]</code>) corresponds to the requested | |
| final size (height, width) of each prediction in batch. If unset, predictions will not be resized.`,name:"target_sizes"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr.py#L1957",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of dictionaries, one per image, each dictionary containing two keys:</p> | |
| <ul> | |
| <li><strong>segmentation</strong> — a tensor of shape <code>(height, width)</code> where each pixel represents a <code>segment_id</code> or | |
| <code>None</code> if no mask if found above <code>threshold</code>. If <code>target_sizes</code> is specified, segmentation is resized to | |
| the corresponding <code>target_sizes</code> entry.</li> | |
| <li><strong>segments_info</strong> — A dictionary that contains additional information on each segment.<ul> | |
| <li><strong>id</strong> — an integer representing the <code>segment_id</code>.</li> | |
| <li><strong>label_id</strong> — An integer representing the label / semantic class id corresponding to <code>segment_id</code>.</li> | |
| <li><strong>was_fused</strong> — a boolean, <code>True</code> if <code>label_id</code> was in <code>label_ids_to_fuse</code>, <code>False</code> otherwise. | |
| Multiple instances of the same class / label were fused and assigned a single <code>segment_id</code>.</li> | |
| <li><strong>score</strong> — Prediction score of segment with <code>segment_id</code>.</li> | |
| </ul></li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[Dict]</code></p> | |
| `}}),st=new J({props:{title:"DetrImageProcessorFast",local:"transformers.DetrImageProcessorFast",headingTag:"h2"}}),rt=new w({props:{name:"class transformers.DetrImageProcessorFast",anchor:"transformers.DetrImageProcessorFast",parameters:[{name:"**kwargs",val:": typing_extensions.Unpack[transformers.models.detr.image_processing_detr_fast.DetrFastImageProcessorKwargs]"}],parametersDescription:[{anchor:"transformers.DetrImageProcessorFast.do_resize",description:`<strong>do_resize</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_resize</code>) — | |
| Whether to resize the image’s (height, width) dimensions to the specified <code>size</code>. Can be overridden by the | |
| <code>do_resize</code> parameter in the <code>preprocess</code> method.`,name:"do_resize"},{anchor:"transformers.DetrImageProcessorFast.size",description:`<strong>size</strong> (<code>dict</code>, <em>optional</em>, defaults to <code>self.size</code>) — | |
| Size of the output image after resizing. Can be overridden by the <code>size</code> parameter in the <code>preprocess</code> | |
| method.`,name:"size"},{anchor:"transformers.DetrImageProcessorFast.default_to_square",description:`<strong>default_to_square</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.default_to_square</code>) — | |
| Whether to default to a square image when resizing, if size is an int.`,name:"default_to_square"},{anchor:"transformers.DetrImageProcessorFast.resample",description:`<strong>resample</strong> (<code>PILImageResampling</code>, <em>optional</em>, defaults to <code>self.resample</code>) — | |
| Resampling filter to use if resizing the image. Only has an effect if <code>do_resize</code> is set to <code>True</code>. Can be | |
| overridden by the <code>resample</code> parameter in the <code>preprocess</code> method.`,name:"resample"},{anchor:"transformers.DetrImageProcessorFast.do_center_crop",description:`<strong>do_center_crop</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_center_crop</code>) — | |
| Whether to center crop the image to the specified <code>crop_size</code>. Can be overridden by <code>do_center_crop</code> in the | |
| <code>preprocess</code> method.`,name:"do_center_crop"},{anchor:"transformers.DetrImageProcessorFast.crop_size",description:`<strong>crop_size</strong> (<code>Dict[str, int]</code> <em>optional</em>, defaults to <code>self.crop_size</code>) — | |
| Size of the output image after applying <code>center_crop</code>. Can be overridden by <code>crop_size</code> in the <code>preprocess</code> | |
| method.`,name:"crop_size"},{anchor:"transformers.DetrImageProcessorFast.do_rescale",description:`<strong>do_rescale</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_rescale</code>) — | |
| Whether to rescale the image by the specified scale <code>rescale_factor</code>. Can be overridden by the | |
| <code>do_rescale</code> parameter in the <code>preprocess</code> method.`,name:"do_rescale"},{anchor:"transformers.DetrImageProcessorFast.rescale_factor",description:`<strong>rescale_factor</strong> (<code>int</code> or <code>float</code>, <em>optional</em>, defaults to <code>self.rescale_factor</code>) — | |
| Scale factor to use if rescaling the image. Only has an effect if <code>do_rescale</code> is set to <code>True</code>. Can be | |
| overridden by the <code>rescale_factor</code> parameter in the <code>preprocess</code> method.`,name:"rescale_factor"},{anchor:"transformers.DetrImageProcessorFast.do_normalize",description:`<strong>do_normalize</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_normalize</code>) — | |
| Whether to normalize the image. Can be overridden by the <code>do_normalize</code> parameter in the <code>preprocess</code> | |
| method. Can be overridden by the <code>do_normalize</code> parameter in the <code>preprocess</code> method.`,name:"do_normalize"},{anchor:"transformers.DetrImageProcessorFast.image_mean",description:`<strong>image_mean</strong> (<code>float</code> or <code>List[float]</code>, <em>optional</em>, defaults to <code>self.image_mean</code>) — | |
| Mean to use if normalizing the image. This is a float or list of floats the length of the number of | |
| channels in the image. Can be overridden by the <code>image_mean</code> parameter in the <code>preprocess</code> method. Can be | |
| overridden by the <code>image_mean</code> parameter in the <code>preprocess</code> method.`,name:"image_mean"},{anchor:"transformers.DetrImageProcessorFast.image_std",description:`<strong>image_std</strong> (<code>float</code> or <code>List[float]</code>, <em>optional</em>, defaults to <code>self.image_std</code>) — | |
| Standard deviation to use if normalizing the image. This is a float or list of floats the length of the | |
| number of channels in the image. Can be overridden by the <code>image_std</code> parameter in the <code>preprocess</code> method. | |
| Can be overridden by the <code>image_std</code> parameter in the <code>preprocess</code> method.`,name:"image_std"},{anchor:"transformers.DetrImageProcessorFast.do_convert_rgb",description:`<strong>do_convert_rgb</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_convert_rgb</code>) — | |
| Whether to convert the image to RGB.`,name:"do_convert_rgb"},{anchor:"transformers.DetrImageProcessorFast.return_tensors",description:"<strong>return_tensors</strong> (<code>str</code> or <code>TensorType</code>, <em>optional</em>, defaults to <code>self.return_tensors</code>) —\nReturns stacked tensors if set to `pt, otherwise returns a list of tensors.",name:"return_tensors"},{anchor:"transformers.DetrImageProcessorFast.data_format",description:`<strong>data_format</strong> (<code>ChannelDimension</code> or <code>str</code>, <em>optional</em>, defaults to <code>self.data_format</code>) — | |
| Only <code>ChannelDimension.FIRST</code> is supported. Added for compatibility with slow processors.`,name:"data_format"},{anchor:"transformers.DetrImageProcessorFast.input_data_format",description:`<strong>input_data_format</strong> (<code>ChannelDimension</code> or <code>str</code>, <em>optional</em>, defaults to <code>self.input_data_format</code>) — | |
| The channel dimension format for the input image. If unset, the channel dimension format is inferred | |
| from the input image. Can be one of:<ul> | |
| <li><code>"channels_first"</code> or <code>ChannelDimension.FIRST</code>: image in (num_channels, height, width) format.</li> | |
| <li><code>"channels_last"</code> or <code>ChannelDimension.LAST</code>: image in (height, width, num_channels) format.</li> | |
| <li><code>"none"</code> or <code>ChannelDimension.NONE</code>: image in (height, width) format.</li> | |
| </ul>`,name:"input_data_format"},{anchor:"transformers.DetrImageProcessorFast.device",description:`<strong>device</strong> (<code>torch.device</code>, <em>optional</em>, defaults to <code>self.device</code>) — | |
| The device to process the images on. If unset, the device is inferred from the input images.`,name:"device"},{anchor:"transformers.DetrImageProcessorFast.format",description:`<strong>format</strong> (<code>str</code>, <em>optional</em>, defaults to <code>AnnotationFormat.COCO_DETECTION</code>) — | |
| Data format of the annotations. One of “coco_detection” or “coco_panoptic”.`,name:"format"},{anchor:"transformers.DetrImageProcessorFast.do_convert_annotations",description:`<strong>do_convert_annotations</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Controls whether to convert the annotations to the format expected by the DETR model. Converts the | |
| bounding boxes to the format <code>(center_x, center_y, width, height)</code> and in the range <code>[0, 1]</code>. | |
| Can be overridden by the <code>do_convert_annotations</code> parameter in the <code>preprocess</code> method.`,name:"do_convert_annotations"},{anchor:"transformers.DetrImageProcessorFast.do_pad",description:`<strong>do_pad</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Controls whether to pad the image. Can be overridden by the <code>do_pad</code> parameter in the <code>preprocess</code> | |
| method. If <code>True</code>, padding will be applied to the bottom and right of the image with zeros. | |
| If <code>pad_size</code> is provided, the image will be padded to the specified dimensions. | |
| Otherwise, the image will be padded to the maximum height and width of the batch.`,name:"do_pad"},{anchor:"transformers.DetrImageProcessorFast.pad_size",description:`<strong>pad_size</strong> (<code>Dict[str, int]</code>, <em>optional</em>) — | |
| The size <code>{"height": int, "width" int}</code> to pad the images to. Must be larger than any image size | |
| provided for preprocessing. If <code>pad_size</code> is not provided, images will be padded to the largest | |
| height and width in the batch.`,name:"pad_size"},{anchor:"transformers.DetrImageProcessorFast.return_segmentation_masks",description:`<strong>return_segmentation_masks</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to return segmentation masks.`,name:"return_segmentation_masks"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr_fast.py#L293"}}),at=new w({props:{name:"preprocess",anchor:"transformers.DetrImageProcessorFast.preprocess",parameters:[{name:"images",val:": typing.Union[ForwardRef('PIL.Image.Image'), numpy.ndarray, ForwardRef('torch.Tensor'), typing.List[ForwardRef('PIL.Image.Image')], typing.List[numpy.ndarray], typing.List[ForwardRef('torch.Tensor')]]"},{name:"annotations",val:": typing.Union[typing.Dict[str, typing.Union[int, str, typing.List[typing.Dict]]], typing.List[typing.Dict[str, typing.Union[int, str, typing.List[typing.Dict]]]], NoneType] = None"},{name:"masks_path",val:": typing.Union[str, pathlib.Path, NoneType] = None"},{name:"**kwargs",val:": typing_extensions.Unpack[transformers.models.detr.image_processing_detr_fast.DetrFastImageProcessorKwargs]"}],parametersDescription:[{anchor:"transformers.DetrImageProcessorFast.preprocess.images",description:`<strong>images</strong> (<code>ImageInput</code>) — | |
| Image to preprocess. Expects a single or batch of images with pixel values ranging from 0 to 255. If | |
| passing in images with pixel values between 0 and 1, set <code>do_rescale=False</code>.`,name:"images"},{anchor:"transformers.DetrImageProcessorFast.preprocess.do_resize",description:`<strong>do_resize</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_resize</code>) — | |
| Whether to resize the image.`,name:"do_resize"},{anchor:"transformers.DetrImageProcessorFast.preprocess.size",description:`<strong>size</strong> (<code>Dict[str, int]</code>, <em>optional</em>, defaults to <code>self.size</code>) — | |
| Describes the maximum input dimensions to the model.`,name:"size"},{anchor:"transformers.DetrImageProcessorFast.preprocess.resample",description:`<strong>resample</strong> (<code>PILImageResampling</code> or <code>InterpolationMode</code>, <em>optional</em>, defaults to <code>self.resample</code>) — | |
| Resampling filter to use if resizing the image. This can be one of the enum <code>PILImageResampling</code>. Only | |
| has an effect if <code>do_resize</code> is set to <code>True</code>.`,name:"resample"},{anchor:"transformers.DetrImageProcessorFast.preprocess.do_center_crop",description:`<strong>do_center_crop</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_center_crop</code>) — | |
| Whether to center crop the image.`,name:"do_center_crop"},{anchor:"transformers.DetrImageProcessorFast.preprocess.crop_size",description:`<strong>crop_size</strong> (<code>Dict[str, int]</code>, <em>optional</em>, defaults to <code>self.crop_size</code>) — | |
| Size of the output image after applying <code>center_crop</code>.`,name:"crop_size"},{anchor:"transformers.DetrImageProcessorFast.preprocess.do_rescale",description:`<strong>do_rescale</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_rescale</code>) — | |
| Whether to rescale the image.`,name:"do_rescale"},{anchor:"transformers.DetrImageProcessorFast.preprocess.rescale_factor",description:`<strong>rescale_factor</strong> (<code>float</code>, <em>optional</em>, defaults to <code>self.rescale_factor</code>) — | |
| Rescale factor to rescale the image by if <code>do_rescale</code> is set to <code>True</code>.`,name:"rescale_factor"},{anchor:"transformers.DetrImageProcessorFast.preprocess.do_normalize",description:`<strong>do_normalize</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_normalize</code>) — | |
| Whether to normalize the image.`,name:"do_normalize"},{anchor:"transformers.DetrImageProcessorFast.preprocess.image_mean",description:`<strong>image_mean</strong> (<code>float</code> or <code>List[float]</code>, <em>optional</em>, defaults to <code>self.image_mean</code>) — | |
| Image mean to use for normalization. Only has an effect if <code>do_normalize</code> is set to <code>True</code>.`,name:"image_mean"},{anchor:"transformers.DetrImageProcessorFast.preprocess.image_std",description:`<strong>image_std</strong> (<code>float</code> or <code>List[float]</code>, <em>optional</em>, defaults to <code>self.image_std</code>) — | |
| Image standard deviation to use for normalization. Only has an effect if <code>do_normalize</code> is set to | |
| <code>True</code>.`,name:"image_std"},{anchor:"transformers.DetrImageProcessorFast.preprocess.do_convert_rgb",description:`<strong>do_convert_rgb</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>self.do_convert_rgb</code>) — | |
| Whether to convert the image to RGB.`,name:"do_convert_rgb"},{anchor:"transformers.DetrImageProcessorFast.preprocess.return_tensors",description:"<strong>return_tensors</strong> (<code>str</code> or <code>TensorType</code>, <em>optional</em>, defaults to <code>self.return_tensors</code>) —\nReturns stacked tensors if set to `pt, otherwise returns a list of tensors.",name:"return_tensors"},{anchor:"transformers.DetrImageProcessorFast.preprocess.data_format",description:`<strong>data_format</strong> (<code>ChannelDimension</code> or <code>str</code>, <em>optional</em>, defaults to <code>self.data_format</code>) — | |
| Only <code>ChannelDimension.FIRST</code> is supported. Added for compatibility with slow processors.`,name:"data_format"},{anchor:"transformers.DetrImageProcessorFast.preprocess.input_data_format",description:`<strong>input_data_format</strong> (<code>ChannelDimension</code> or <code>str</code>, <em>optional</em>, defaults to <code>self.input_data_format</code>) — | |
| The channel dimension format for the input image. If unset, the channel dimension format is inferred | |
| from the input image. Can be one of:<ul> | |
| <li><code>"channels_first"</code> or <code>ChannelDimension.FIRST</code>: image in (num_channels, height, width) format.</li> | |
| <li><code>"channels_last"</code> or <code>ChannelDimension.LAST</code>: image in (height, width, num_channels) format.</li> | |
| <li><code>"none"</code> or <code>ChannelDimension.NONE</code>: image in (height, width) format.</li> | |
| </ul>`,name:"input_data_format"},{anchor:"transformers.DetrImageProcessorFast.preprocess.device",description:`<strong>device</strong> (<code>torch.device</code>, <em>optional</em>, defaults to <code>self.device</code>) — | |
| The device to process the images on. If unset, the device is inferred from the input images.`,name:"device"},{anchor:"transformers.DetrImageProcessorFast.preprocess.annotations",description:`<strong>annotations</strong> (<code>AnnotationType</code> or <code>List[AnnotationType]</code>, <em>optional</em>) — | |
| List of annotations associated with the image or batch of images. If annotation is for object | |
| detection, the annotations should be a dictionary with the following keys:<ul> | |
| <li>“image_id” (<code>int</code>): The image id.</li> | |
| <li>“annotations” (<code>List[Dict]</code>): List of annotations for an image. Each annotation should be a | |
| dictionary. An image can have no annotations, in which case the list should be empty. | |
| If annotation is for segmentation, the annotations should be a dictionary with the following keys:</li> | |
| <li>“image_id” (<code>int</code>): The image id.</li> | |
| <li>“segments_info” (<code>List[Dict]</code>): List of segments for an image. Each segment should be a dictionary. | |
| An image can have no segments, in which case the list should be empty.</li> | |
| <li>“file_name” (<code>str</code>): The file name of the image.</li> | |
| </ul>`,name:"annotations"},{anchor:"transformers.DetrImageProcessorFast.preprocess.format",description:`<strong>format</strong> (<code>str</code>, <em>optional</em>, defaults to <code>AnnotationFormat.COCO_DETECTION</code>) — | |
| Data format of the annotations. One of “coco_detection” or “coco_panoptic”.`,name:"format"},{anchor:"transformers.DetrImageProcessorFast.preprocess.do_convert_annotations",description:`<strong>do_convert_annotations</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Controls whether to convert the annotations to the format expected by the DETR model. Converts the | |
| bounding boxes to the format <code>(center_x, center_y, width, height)</code> and in the range <code>[0, 1]</code>. | |
| Can be overridden by the <code>do_convert_annotations</code> parameter in the <code>preprocess</code> method.`,name:"do_convert_annotations"},{anchor:"transformers.DetrImageProcessorFast.preprocess.do_pad",description:`<strong>do_pad</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Controls whether to pad the image. Can be overridden by the <code>do_pad</code> parameter in the <code>preprocess</code> | |
| method. If <code>True</code>, padding will be applied to the bottom and right of the image with zeros. | |
| If <code>pad_size</code> is provided, the image will be padded to the specified dimensions. | |
| Otherwise, the image will be padded to the maximum height and width of the batch.`,name:"do_pad"},{anchor:"transformers.DetrImageProcessorFast.preprocess.pad_size",description:`<strong>pad_size</strong> (<code>Dict[str, int]</code>, <em>optional</em>) — | |
| The size <code>{"height": int, "width" int}</code> to pad the images to. Must be larger than any image size | |
| provided for preprocessing. If <code>pad_size</code> is not provided, images will be padded to the largest | |
| height and width in the batch.`,name:"pad_size"},{anchor:"transformers.DetrImageProcessorFast.preprocess.return_segmentation_masks",description:`<strong>return_segmentation_masks</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to return segmentation masks.`,name:"return_segmentation_masks"},{anchor:"transformers.DetrImageProcessorFast.preprocess.masks_path",description:`<strong>masks_path</strong> (<code>str</code> or <code>pathlib.Path</code>, <em>optional</em>) — | |
| Path to the directory containing the segmentation masks.`,name:"masks_path"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr_fast.py#L588"}}),it=new w({props:{name:"post_process_object_detection",anchor:"transformers.DetrImageProcessorFast.post_process_object_detection",parameters:[{name:"outputs",val:""},{name:"threshold",val:": float = 0.5"},{name:"target_sizes",val:": typing.Union[transformers.utils.generic.TensorType, typing.List[typing.Tuple]] = None"}],parametersDescription:[{anchor:"transformers.DetrImageProcessorFast.post_process_object_detection.outputs",description:`<strong>outputs</strong> (<code>DetrObjectDetectionOutput</code>) — | |
| Raw outputs of the model.`,name:"outputs"},{anchor:"transformers.DetrImageProcessorFast.post_process_object_detection.threshold",description:`<strong>threshold</strong> (<code>float</code>, <em>optional</em>) — | |
| Score threshold to keep object detection predictions.`,name:"threshold"},{anchor:"transformers.DetrImageProcessorFast.post_process_object_detection.target_sizes",description:`<strong>target_sizes</strong> (<code>torch.Tensor</code> or <code>List[Tuple[int, int]]</code>, <em>optional</em>) — | |
| Tensor of shape <code>(batch_size, 2)</code> or list of tuples (<code>Tuple[int, int]</code>) containing the target size | |
| <code>(height, width)</code> of each image in the batch. If unset, predictions will not be resized.`,name:"target_sizes"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr_fast.py#L1035",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of dictionaries, each dictionary containing the scores, labels and boxes for an image | |
| in the batch as predicted by the model.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[Dict]</code></p> | |
| `}}),dt=new w({props:{name:"post_process_semantic_segmentation",anchor:"transformers.DetrImageProcessorFast.post_process_semantic_segmentation",parameters:[{name:"outputs",val:""},{name:"target_sizes",val:": typing.List[typing.Tuple[int, int]] = None"}],parametersDescription:[{anchor:"transformers.DetrImageProcessorFast.post_process_semantic_segmentation.outputs",description:`<strong>outputs</strong> (<a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a>) — | |
| Raw outputs of the model.`,name:"outputs"},{anchor:"transformers.DetrImageProcessorFast.post_process_semantic_segmentation.target_sizes",description:`<strong>target_sizes</strong> (<code>List[Tuple[int, int]]</code>, <em>optional</em>) — | |
| A list of tuples (<code>Tuple[int, int]</code>) containing the target size (height, width) of each image in the | |
| batch. If unset, predictions will not be resized.`,name:"target_sizes"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr_fast.py#L1089",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of length <code>batch_size</code>, where each item is a semantic segmentation map of shape (height, width) | |
| corresponding to the target_sizes entry (if <code>target_sizes</code> is specified). Each entry of each | |
| <code>torch.Tensor</code> correspond to a semantic class id.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[torch.Tensor]</code></p> | |
| `}}),ct=new w({props:{name:"post_process_instance_segmentation",anchor:"transformers.DetrImageProcessorFast.post_process_instance_segmentation",parameters:[{name:"outputs",val:""},{name:"threshold",val:": float = 0.5"},{name:"mask_threshold",val:": float = 0.5"},{name:"overlap_mask_area_threshold",val:": float = 0.8"},{name:"target_sizes",val:": typing.Optional[typing.List[typing.Tuple[int, int]]] = None"},{name:"return_coco_annotation",val:": typing.Optional[bool] = False"}],parametersDescription:[{anchor:"transformers.DetrImageProcessorFast.post_process_instance_segmentation.outputs",description:`<strong>outputs</strong> (<a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a>) — | |
| Raw outputs of the model.`,name:"outputs"},{anchor:"transformers.DetrImageProcessorFast.post_process_instance_segmentation.threshold",description:`<strong>threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.5) — | |
| The probability score threshold to keep predicted instance masks.`,name:"threshold"},{anchor:"transformers.DetrImageProcessorFast.post_process_instance_segmentation.mask_threshold",description:`<strong>mask_threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.5) — | |
| Threshold to use when turning the predicted masks into binary values.`,name:"mask_threshold"},{anchor:"transformers.DetrImageProcessorFast.post_process_instance_segmentation.overlap_mask_area_threshold",description:`<strong>overlap_mask_area_threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.8) — | |
| The overlap mask area threshold to merge or discard small disconnected parts within each binary | |
| instance mask.`,name:"overlap_mask_area_threshold"},{anchor:"transformers.DetrImageProcessorFast.post_process_instance_segmentation.target_sizes",description:`<strong>target_sizes</strong> (<code>List[Tuple]</code>, <em>optional</em>) — | |
| List of length (batch_size), where each list item (<code>Tuple[int, int]]</code>) corresponds to the requested | |
| final size (height, width) of each prediction. If unset, predictions will not be resized.`,name:"target_sizes"},{anchor:"transformers.DetrImageProcessorFast.post_process_instance_segmentation.return_coco_annotation",description:`<strong>return_coco_annotation</strong> (<code>bool</code>, <em>optional</em>) — | |
| Defaults to <code>False</code>. If set to <code>True</code>, segmentation maps are returned in COCO run-length encoding (RLE) | |
| format.`,name:"return_coco_annotation"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr_fast.py#L1137",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of dictionaries, one per image, each dictionary containing two keys:</p> | |
| <ul> | |
| <li><strong>segmentation</strong> — A tensor of shape <code>(height, width)</code> where each pixel represents a <code>segment_id</code> or | |
| <code>List[List]</code> run-length encoding (RLE) of the segmentation map if return_coco_annotation is set to | |
| <code>True</code>. Set to <code>None</code> if no mask if found above <code>threshold</code>.</li> | |
| <li><strong>segments_info</strong> — A dictionary that contains additional information on each segment.<ul> | |
| <li><strong>id</strong> — An integer representing the <code>segment_id</code>.</li> | |
| <li><strong>label_id</strong> — An integer representing the label / semantic class id corresponding to <code>segment_id</code>.</li> | |
| <li><strong>score</strong> — Prediction score of segment with <code>segment_id</code>.</li> | |
| </ul></li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[Dict]</code></p> | |
| `}}),lt=new w({props:{name:"post_process_panoptic_segmentation",anchor:"transformers.DetrImageProcessorFast.post_process_panoptic_segmentation",parameters:[{name:"outputs",val:""},{name:"threshold",val:": float = 0.5"},{name:"mask_threshold",val:": float = 0.5"},{name:"overlap_mask_area_threshold",val:": float = 0.8"},{name:"label_ids_to_fuse",val:": typing.Optional[typing.Set[int]] = None"},{name:"target_sizes",val:": typing.Optional[typing.List[typing.Tuple[int, int]]] = None"}],parametersDescription:[{anchor:"transformers.DetrImageProcessorFast.post_process_panoptic_segmentation.outputs",description:`<strong>outputs</strong> (<a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a>) — | |
| The outputs from <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a>.`,name:"outputs"},{anchor:"transformers.DetrImageProcessorFast.post_process_panoptic_segmentation.threshold",description:`<strong>threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.5) — | |
| The probability score threshold to keep predicted instance masks.`,name:"threshold"},{anchor:"transformers.DetrImageProcessorFast.post_process_panoptic_segmentation.mask_threshold",description:`<strong>mask_threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.5) — | |
| Threshold to use when turning the predicted masks into binary values.`,name:"mask_threshold"},{anchor:"transformers.DetrImageProcessorFast.post_process_panoptic_segmentation.overlap_mask_area_threshold",description:`<strong>overlap_mask_area_threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.8) — | |
| The overlap mask area threshold to merge or discard small disconnected parts within each binary | |
| instance mask.`,name:"overlap_mask_area_threshold"},{anchor:"transformers.DetrImageProcessorFast.post_process_panoptic_segmentation.label_ids_to_fuse",description:`<strong>label_ids_to_fuse</strong> (<code>Set[int]</code>, <em>optional</em>) — | |
| The labels in this state will have all their instances be fused together. For instance we could say | |
| there can only be one sky in an image, but several persons, so the label ID for sky would be in that | |
| set, but not the one for person.`,name:"label_ids_to_fuse"},{anchor:"transformers.DetrImageProcessorFast.post_process_panoptic_segmentation.target_sizes",description:`<strong>target_sizes</strong> (<code>List[Tuple]</code>, <em>optional</em>) — | |
| List of length (batch_size), where each list item (<code>Tuple[int, int]]</code>) corresponds to the requested | |
| final size (height, width) of each prediction in batch. If unset, predictions will not be resized.`,name:"target_sizes"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr_fast.py#L1221",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of dictionaries, one per image, each dictionary containing two keys:</p> | |
| <ul> | |
| <li><strong>segmentation</strong> — a tensor of shape <code>(height, width)</code> where each pixel represents a <code>segment_id</code> or | |
| <code>None</code> if no mask if found above <code>threshold</code>. If <code>target_sizes</code> is specified, segmentation is resized to | |
| the corresponding <code>target_sizes</code> entry.</li> | |
| <li><strong>segments_info</strong> — A dictionary that contains additional information on each segment.<ul> | |
| <li><strong>id</strong> — an integer representing the <code>segment_id</code>.</li> | |
| <li><strong>label_id</strong> — An integer representing the label / semantic class id corresponding to <code>segment_id</code>.</li> | |
| <li><strong>was_fused</strong> — a boolean, <code>True</code> if <code>label_id</code> was in <code>label_ids_to_fuse</code>, <code>False</code> otherwise. | |
| Multiple instances of the same class / label were fused and assigned a single <code>segment_id</code>.</li> | |
| <li><strong>score</strong> — Prediction score of segment with <code>segment_id</code>.</li> | |
| </ul></li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[Dict]</code></p> | |
| `}}),mt=new J({props:{title:"DetrFeatureExtractor",local:"transformers.DetrFeatureExtractor",headingTag:"h2"}}),pt=new w({props:{name:"class transformers.DetrFeatureExtractor",anchor:"transformers.DetrFeatureExtractor",parameters:[{name:"*args",val:""},{name:"**kwargs",val:""}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/feature_extraction_detr.py#L36"}}),ht=new w({props:{name:"__call__",anchor:"transformers.DetrFeatureExtractor.__call__",parameters:[{name:"images",val:""},{name:"**kwargs",val:""}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/image_processing_utils.py#L40"}}),gt=new w({props:{name:"post_process_object_detection",anchor:"transformers.DetrFeatureExtractor.post_process_object_detection",parameters:[{name:"outputs",val:""},{name:"threshold",val:": float = 0.5"},{name:"target_sizes",val:": typing.Union[transformers.utils.generic.TensorType, typing.List[typing.Tuple]] = None"}],parametersDescription:[{anchor:"transformers.DetrFeatureExtractor.post_process_object_detection.outputs",description:`<strong>outputs</strong> (<code>DetrObjectDetectionOutput</code>) — | |
| Raw outputs of the model.`,name:"outputs"},{anchor:"transformers.DetrFeatureExtractor.post_process_object_detection.threshold",description:`<strong>threshold</strong> (<code>float</code>, <em>optional</em>) — | |
| Score threshold to keep object detection predictions.`,name:"threshold"},{anchor:"transformers.DetrFeatureExtractor.post_process_object_detection.target_sizes",description:`<strong>target_sizes</strong> (<code>torch.Tensor</code> or <code>List[Tuple[int, int]]</code>, <em>optional</em>) — | |
| Tensor of shape <code>(batch_size, 2)</code> or list of tuples (<code>Tuple[int, int]</code>) containing the target size | |
| <code>(height, width)</code> of each image in the batch. If unset, predictions will not be resized.`,name:"target_sizes"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr.py#L1772",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of dictionaries, each dictionary containing the scores, labels and boxes for an image | |
| in the batch as predicted by the model.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[Dict]</code></p> | |
| `}}),ft=new w({props:{name:"post_process_semantic_segmentation",anchor:"transformers.DetrFeatureExtractor.post_process_semantic_segmentation",parameters:[{name:"outputs",val:""},{name:"target_sizes",val:": typing.List[typing.Tuple[int, int]] = None"}],parametersDescription:[{anchor:"transformers.DetrFeatureExtractor.post_process_semantic_segmentation.outputs",description:`<strong>outputs</strong> (<a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a>) — | |
| Raw outputs of the model.`,name:"outputs"},{anchor:"transformers.DetrFeatureExtractor.post_process_semantic_segmentation.target_sizes",description:`<strong>target_sizes</strong> (<code>List[Tuple[int, int]]</code>, <em>optional</em>) — | |
| A list of tuples (<code>Tuple[int, int]</code>) containing the target size (height, width) of each image in the | |
| batch. If unset, predictions will not be resized.`,name:"target_sizes"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr.py#L1825",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of length <code>batch_size</code>, where each item is a semantic segmentation map of shape (height, width) | |
| corresponding to the target_sizes entry (if <code>target_sizes</code> is specified). Each entry of each | |
| <code>torch.Tensor</code> correspond to a semantic class id.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[torch.Tensor]</code></p> | |
| `}}),ut=new w({props:{name:"post_process_instance_segmentation",anchor:"transformers.DetrFeatureExtractor.post_process_instance_segmentation",parameters:[{name:"outputs",val:""},{name:"threshold",val:": float = 0.5"},{name:"mask_threshold",val:": float = 0.5"},{name:"overlap_mask_area_threshold",val:": float = 0.8"},{name:"target_sizes",val:": typing.Optional[typing.List[typing.Tuple[int, int]]] = None"},{name:"return_coco_annotation",val:": typing.Optional[bool] = False"}],parametersDescription:[{anchor:"transformers.DetrFeatureExtractor.post_process_instance_segmentation.outputs",description:`<strong>outputs</strong> (<a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a>) — | |
| Raw outputs of the model.`,name:"outputs"},{anchor:"transformers.DetrFeatureExtractor.post_process_instance_segmentation.threshold",description:`<strong>threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.5) — | |
| The probability score threshold to keep predicted instance masks.`,name:"threshold"},{anchor:"transformers.DetrFeatureExtractor.post_process_instance_segmentation.mask_threshold",description:`<strong>mask_threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.5) — | |
| Threshold to use when turning the predicted masks into binary values.`,name:"mask_threshold"},{anchor:"transformers.DetrFeatureExtractor.post_process_instance_segmentation.overlap_mask_area_threshold",description:`<strong>overlap_mask_area_threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.8) — | |
| The overlap mask area threshold to merge or discard small disconnected parts within each binary | |
| instance mask.`,name:"overlap_mask_area_threshold"},{anchor:"transformers.DetrFeatureExtractor.post_process_instance_segmentation.target_sizes",description:`<strong>target_sizes</strong> (<code>List[Tuple]</code>, <em>optional</em>) — | |
| List of length (batch_size), where each list item (<code>Tuple[int, int]]</code>) corresponds to the requested | |
| final size (height, width) of each prediction. If unset, predictions will not be resized.`,name:"target_sizes"},{anchor:"transformers.DetrFeatureExtractor.post_process_instance_segmentation.return_coco_annotation",description:`<strong>return_coco_annotation</strong> (<code>bool</code>, <em>optional</em>) — | |
| Defaults to <code>False</code>. If set to <code>True</code>, segmentation maps are returned in COCO run-length encoding (RLE) | |
| format.`,name:"return_coco_annotation"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr.py#L1873",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of dictionaries, one per image, each dictionary containing two keys:</p> | |
| <ul> | |
| <li><strong>segmentation</strong> — A tensor of shape <code>(height, width)</code> where each pixel represents a <code>segment_id</code> or | |
| <code>List[List]</code> run-length encoding (RLE) of the segmentation map if return_coco_annotation is set to | |
| <code>True</code>. Set to <code>None</code> if no mask if found above <code>threshold</code>.</li> | |
| <li><strong>segments_info</strong> — A dictionary that contains additional information on each segment.<ul> | |
| <li><strong>id</strong> — An integer representing the <code>segment_id</code>.</li> | |
| <li><strong>label_id</strong> — An integer representing the label / semantic class id corresponding to <code>segment_id</code>.</li> | |
| <li><strong>score</strong> — Prediction score of segment with <code>segment_id</code>.</li> | |
| </ul></li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[Dict]</code></p> | |
| `}}),_t=new w({props:{name:"post_process_panoptic_segmentation",anchor:"transformers.DetrFeatureExtractor.post_process_panoptic_segmentation",parameters:[{name:"outputs",val:""},{name:"threshold",val:": float = 0.5"},{name:"mask_threshold",val:": float = 0.5"},{name:"overlap_mask_area_threshold",val:": float = 0.8"},{name:"label_ids_to_fuse",val:": typing.Optional[typing.Set[int]] = None"},{name:"target_sizes",val:": typing.Optional[typing.List[typing.Tuple[int, int]]] = None"}],parametersDescription:[{anchor:"transformers.DetrFeatureExtractor.post_process_panoptic_segmentation.outputs",description:`<strong>outputs</strong> (<a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a>) — | |
| The outputs from <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrForSegmentation">DetrForSegmentation</a>.`,name:"outputs"},{anchor:"transformers.DetrFeatureExtractor.post_process_panoptic_segmentation.threshold",description:`<strong>threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.5) — | |
| The probability score threshold to keep predicted instance masks.`,name:"threshold"},{anchor:"transformers.DetrFeatureExtractor.post_process_panoptic_segmentation.mask_threshold",description:`<strong>mask_threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.5) — | |
| Threshold to use when turning the predicted masks into binary values.`,name:"mask_threshold"},{anchor:"transformers.DetrFeatureExtractor.post_process_panoptic_segmentation.overlap_mask_area_threshold",description:`<strong>overlap_mask_area_threshold</strong> (<code>float</code>, <em>optional</em>, defaults to 0.8) — | |
| The overlap mask area threshold to merge or discard small disconnected parts within each binary | |
| instance mask.`,name:"overlap_mask_area_threshold"},{anchor:"transformers.DetrFeatureExtractor.post_process_panoptic_segmentation.label_ids_to_fuse",description:`<strong>label_ids_to_fuse</strong> (<code>Set[int]</code>, <em>optional</em>) — | |
| The labels in this state will have all their instances be fused together. For instance we could say | |
| there can only be one sky in an image, but several persons, so the label ID for sky would be in that | |
| set, but not the one for person.`,name:"label_ids_to_fuse"},{anchor:"transformers.DetrFeatureExtractor.post_process_panoptic_segmentation.target_sizes",description:`<strong>target_sizes</strong> (<code>List[Tuple]</code>, <em>optional</em>) — | |
| List of length (batch_size), where each list item (<code>Tuple[int, int]]</code>) corresponds to the requested | |
| final size (height, width) of each prediction in batch. If unset, predictions will not be resized.`,name:"target_sizes"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/image_processing_detr.py#L1957",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A list of dictionaries, one per image, each dictionary containing two keys:</p> | |
| <ul> | |
| <li><strong>segmentation</strong> — a tensor of shape <code>(height, width)</code> where each pixel represents a <code>segment_id</code> or | |
| <code>None</code> if no mask if found above <code>threshold</code>. If <code>target_sizes</code> is specified, segmentation is resized to | |
| the corresponding <code>target_sizes</code> entry.</li> | |
| <li><strong>segments_info</strong> — A dictionary that contains additional information on each segment.<ul> | |
| <li><strong>id</strong> — an integer representing the <code>segment_id</code>.</li> | |
| <li><strong>label_id</strong> — An integer representing the label / semantic class id corresponding to <code>segment_id</code>.</li> | |
| <li><strong>was_fused</strong> — a boolean, <code>True</code> if <code>label_id</code> was in <code>label_ids_to_fuse</code>, <code>False</code> otherwise. | |
| Multiple instances of the same class / label were fused and assigned a single <code>segment_id</code>.</li> | |
| <li><strong>score</strong> — Prediction score of segment with <code>segment_id</code>.</li> | |
| </ul></li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>List[Dict]</code></p> | |
| `}}),bt=new J({props:{title:"DETR specific outputs",local:"transformers.models.detr.modeling_detr.DetrModelOutput",headingTag:"h2"}}),yt=new w({props:{name:"class transformers.models.detr.modeling_detr.DetrModelOutput",anchor:"transformers.models.detr.modeling_detr.DetrModelOutput",parameters:[{name:"last_hidden_state",val:": FloatTensor = None"},{name:"past_key_values",val:": typing.Optional[typing.Tuple[typing.Tuple[torch.FloatTensor]]] = None"},{name:"decoder_hidden_states",val:": typing.Optional[typing.Tuple[torch.FloatTensor, ...]] = None"},{name:"decoder_attentions",val:": typing.Optional[typing.Tuple[torch.FloatTensor, ...]] = None"},{name:"cross_attentions",val:": typing.Optional[typing.Tuple[torch.FloatTensor, ...]] = None"},{name:"encoder_last_hidden_state",val:": typing.Optional[torch.FloatTensor] = None"},{name:"encoder_hidden_states",val:": typing.Optional[typing.Tuple[torch.FloatTensor, ...]] = None"},{name:"encoder_attentions",val:": typing.Optional[typing.Tuple[torch.FloatTensor, ...]] = None"},{name:"intermediate_hidden_states",val:": typing.Optional[torch.FloatTensor] = None"}],parametersDescription:[{anchor:"transformers.models.detr.modeling_detr.DetrModelOutput.last_hidden_state",description:`<strong>last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>) — | |
| Sequence of hidden-states at the output of the last layer of the decoder of the model.`,name:"last_hidden_state"},{anchor:"transformers.models.detr.modeling_detr.DetrModelOutput.decoder_hidden_states",description:`<strong>decoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings + one for the output of each layer) of | |
| shape <code>(batch_size, sequence_length, hidden_size)</code>. Hidden-states of the decoder at the output of each | |
| layer plus the initial embedding outputs.`,name:"decoder_hidden_states"},{anchor:"transformers.models.detr.modeling_detr.DetrModelOutput.decoder_attentions",description:`<strong>decoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the decoder, after the attention softmax, used to compute the | |
| weighted average in the self-attention heads.`,name:"decoder_attentions"},{anchor:"transformers.models.detr.modeling_detr.DetrModelOutput.cross_attentions",description:`<strong>cross_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the decoder’s cross-attention layer, after the attention softmax, | |
| used to compute the weighted average in the cross-attention heads.`,name:"cross_attentions"},{anchor:"transformers.models.detr.modeling_detr.DetrModelOutput.encoder_last_hidden_state",description:`<strong>encoder_last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Sequence of hidden-states at the output of the last layer of the encoder of the model.`,name:"encoder_last_hidden_state"},{anchor:"transformers.models.detr.modeling_detr.DetrModelOutput.encoder_hidden_states",description:`<strong>encoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings + one for the output of each layer) of | |
| shape <code>(batch_size, sequence_length, hidden_size)</code>. Hidden-states of the encoder at the output of each | |
| layer plus the initial embedding outputs.`,name:"encoder_hidden_states"},{anchor:"transformers.models.detr.modeling_detr.DetrModelOutput.encoder_attentions",description:`<strong>encoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the encoder, after the attention softmax, used to compute the | |
| weighted average in the self-attention heads.`,name:"encoder_attentions"},{anchor:"transformers.models.detr.modeling_detr.DetrModelOutput.intermediate_hidden_states",description:`<strong>intermediate_hidden_states</strong> (<code>torch.FloatTensor</code> of shape <code>(config.decoder_layers, batch_size, sequence_length, hidden_size)</code>, <em>optional</em>, returned when <code>config.auxiliary_loss=True</code>) — | |
| Intermediate decoder activations, i.e. the output of each decoder layer, each of them gone through a | |
| layernorm.`,name:"intermediate_hidden_states"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/modeling_detr.py#L81"}}),Tt=new w({props:{name:"class transformers.models.detr.modeling_detr.DetrObjectDetectionOutput",anchor:"transformers.models.detr.modeling_detr.DetrObjectDetectionOutput",parameters:[{name:"loss",val:": typing.Optional[torch.FloatTensor] = None"},{name:"loss_dict",val:": typing.Optional[typing.Dict] = None"},{name:"logits",val:": FloatTensor = None"},{name:"pred_boxes",val:": FloatTensor = None"},{name:"auxiliary_outputs",val:": typing.Optional[typing.List[typing.Dict]] = None"},{name:"last_hidden_state",val:": typing.Optional[torch.FloatTensor] = None"},{name:"decoder_hidden_states",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"},{name:"decoder_attentions",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"},{name:"cross_attentions",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"},{name:"encoder_last_hidden_state",val:": typing.Optional[torch.FloatTensor] = None"},{name:"encoder_hidden_states",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"},{name:"encoder_attentions",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"}],parametersDescription:[{anchor:"transformers.models.detr.modeling_detr.DetrObjectDetectionOutput.loss",description:`<strong>loss</strong> (<code>torch.FloatTensor</code> of shape <code>(1,)</code>, <em>optional</em>, returned when <code>labels</code> are provided)) — | |
| Total loss as a linear combination of a negative log-likehood (cross-entropy) for class prediction and a | |
| bounding box loss. The latter is defined as a linear combination of the L1 loss and the generalized | |
| scale-invariant IoU loss.`,name:"loss"},{anchor:"transformers.models.detr.modeling_detr.DetrObjectDetectionOutput.loss_dict",description:`<strong>loss_dict</strong> (<code>Dict</code>, <em>optional</em>) — | |
| A dictionary containing the individual losses. Useful for logging.`,name:"loss_dict"},{anchor:"transformers.models.detr.modeling_detr.DetrObjectDetectionOutput.logits",description:`<strong>logits</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, num_classes + 1)</code>) — | |
| Classification logits (including no-object) for all queries.`,name:"logits"},{anchor:"transformers.models.detr.modeling_detr.DetrObjectDetectionOutput.pred_boxes",description:`<strong>pred_boxes</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, 4)</code>) — | |
| Normalized boxes coordinates for all queries, represented as (center_x, center_y, width, height). These | |
| values are normalized in [0, 1], relative to the size of each individual image in the batch (disregarding | |
| possible padding). You can use <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor.post_process_object_detection">post_process_object_detection()</a> to retrieve the | |
| unnormalized bounding boxes.`,name:"pred_boxes"},{anchor:"transformers.models.detr.modeling_detr.DetrObjectDetectionOutput.auxiliary_outputs",description:`<strong>auxiliary_outputs</strong> (<code>list[Dict]</code>, <em>optional</em>) — | |
| Optional, only returned when auxilary losses are activated (i.e. <code>config.auxiliary_loss</code> is set to <code>True</code>) | |
| and labels are provided. It is a list of dictionaries containing the two above keys (<code>logits</code> and | |
| <code>pred_boxes</code>) for each decoder layer.`,name:"auxiliary_outputs"},{anchor:"transformers.models.detr.modeling_detr.DetrObjectDetectionOutput.last_hidden_state",description:`<strong>last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Sequence of hidden-states at the output of the last layer of the decoder of the model.`,name:"last_hidden_state"},{anchor:"transformers.models.detr.modeling_detr.DetrObjectDetectionOutput.decoder_hidden_states",description:`<strong>decoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings + one for the output of each layer) of | |
| shape <code>(batch_size, sequence_length, hidden_size)</code>. Hidden-states of the decoder at the output of each | |
| layer plus the initial embedding outputs.`,name:"decoder_hidden_states"},{anchor:"transformers.models.detr.modeling_detr.DetrObjectDetectionOutput.decoder_attentions",description:`<strong>decoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the decoder, after the attention softmax, used to compute the | |
| weighted average in the self-attention heads.`,name:"decoder_attentions"},{anchor:"transformers.models.detr.modeling_detr.DetrObjectDetectionOutput.cross_attentions",description:`<strong>cross_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the decoder’s cross-attention layer, after the attention softmax, | |
| used to compute the weighted average in the cross-attention heads.`,name:"cross_attentions"},{anchor:"transformers.models.detr.modeling_detr.DetrObjectDetectionOutput.encoder_last_hidden_state",description:`<strong>encoder_last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Sequence of hidden-states at the output of the last layer of the encoder of the model.`,name:"encoder_last_hidden_state"},{anchor:"transformers.models.detr.modeling_detr.DetrObjectDetectionOutput.encoder_hidden_states",description:`<strong>encoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings + one for the output of each layer) of | |
| shape <code>(batch_size, sequence_length, hidden_size)</code>. Hidden-states of the encoder at the output of each | |
| layer plus the initial embedding outputs.`,name:"encoder_hidden_states"},{anchor:"transformers.models.detr.modeling_detr.DetrObjectDetectionOutput.encoder_attentions",description:`<strong>encoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the encoder, after the attention softmax, used to compute the | |
| weighted average in the self-attention heads.`,name:"encoder_attentions"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/modeling_detr.py#L121"}}),vt=new w({props:{name:"class transformers.models.detr.modeling_detr.DetrSegmentationOutput",anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput",parameters:[{name:"loss",val:": typing.Optional[torch.FloatTensor] = None"},{name:"loss_dict",val:": typing.Optional[typing.Dict] = None"},{name:"logits",val:": FloatTensor = None"},{name:"pred_boxes",val:": FloatTensor = None"},{name:"pred_masks",val:": FloatTensor = None"},{name:"auxiliary_outputs",val:": typing.Optional[typing.List[typing.Dict]] = None"},{name:"last_hidden_state",val:": typing.Optional[torch.FloatTensor] = None"},{name:"decoder_hidden_states",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"},{name:"decoder_attentions",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"},{name:"cross_attentions",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"},{name:"encoder_last_hidden_state",val:": typing.Optional[torch.FloatTensor] = None"},{name:"encoder_hidden_states",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"},{name:"encoder_attentions",val:": typing.Optional[typing.Tuple[torch.FloatTensor]] = None"}],parametersDescription:[{anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput.loss",description:`<strong>loss</strong> (<code>torch.FloatTensor</code> of shape <code>(1,)</code>, <em>optional</em>, returned when <code>labels</code> are provided)) — | |
| Total loss as a linear combination of a negative log-likehood (cross-entropy) for class prediction and a | |
| bounding box loss. The latter is defined as a linear combination of the L1 loss and the generalized | |
| scale-invariant IoU loss.`,name:"loss"},{anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput.loss_dict",description:`<strong>loss_dict</strong> (<code>Dict</code>, <em>optional</em>) — | |
| A dictionary containing the individual losses. Useful for logging.`,name:"loss_dict"},{anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput.logits",description:`<strong>logits</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, num_classes + 1)</code>) — | |
| Classification logits (including no-object) for all queries.`,name:"logits"},{anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput.pred_boxes",description:`<strong>pred_boxes</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, 4)</code>) — | |
| Normalized boxes coordinates for all queries, represented as (center_x, center_y, width, height). These | |
| values are normalized in [0, 1], relative to the size of each individual image in the batch (disregarding | |
| possible padding). You can use <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor.post_process_object_detection">post_process_object_detection()</a> to retrieve the | |
| unnormalized bounding boxes.`,name:"pred_boxes"},{anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput.pred_masks",description:`<strong>pred_masks</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, height/4, width/4)</code>) — | |
| Segmentation masks logits for all queries. See also | |
| <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor.post_process_semantic_segmentation">post_process_semantic_segmentation()</a> or | |
| <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor.post_process_instance_segmentation">post_process_instance_segmentation()</a> | |
| <a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor.post_process_panoptic_segmentation">post_process_panoptic_segmentation()</a> to evaluate semantic, instance and panoptic | |
| segmentation masks respectively.`,name:"pred_masks"},{anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput.auxiliary_outputs",description:`<strong>auxiliary_outputs</strong> (<code>list[Dict]</code>, <em>optional</em>) — | |
| Optional, only returned when auxiliary losses are activated (i.e. <code>config.auxiliary_loss</code> is set to <code>True</code>) | |
| and labels are provided. It is a list of dictionaries containing the two above keys (<code>logits</code> and | |
| <code>pred_boxes</code>) for each decoder layer.`,name:"auxiliary_outputs"},{anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput.last_hidden_state",description:`<strong>last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Sequence of hidden-states at the output of the last layer of the decoder of the model.`,name:"last_hidden_state"},{anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput.decoder_hidden_states",description:`<strong>decoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings + one for the output of each layer) of | |
| shape <code>(batch_size, sequence_length, hidden_size)</code>. Hidden-states of the decoder at the output of each | |
| layer plus the initial embedding outputs.`,name:"decoder_hidden_states"},{anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput.decoder_attentions",description:`<strong>decoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the decoder, after the attention softmax, used to compute the | |
| weighted average in the self-attention heads.`,name:"decoder_attentions"},{anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput.cross_attentions",description:`<strong>cross_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the decoder’s cross-attention layer, after the attention softmax, | |
| used to compute the weighted average in the cross-attention heads.`,name:"cross_attentions"},{anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput.encoder_last_hidden_state",description:`<strong>encoder_last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Sequence of hidden-states at the output of the last layer of the encoder of the model.`,name:"encoder_last_hidden_state"},{anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput.encoder_hidden_states",description:`<strong>encoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings + one for the output of each layer) of | |
| shape <code>(batch_size, sequence_length, hidden_size)</code>. Hidden-states of the encoder at the output of each | |
| layer plus the initial embedding outputs.`,name:"encoder_hidden_states"},{anchor:"transformers.models.detr.modeling_detr.DetrSegmentationOutput.encoder_attentions",description:`<strong>encoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — | |
| Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the encoder, after the attention softmax, used to compute the | |
| weighted average in the self-attention heads.`,name:"encoder_attentions"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/modeling_detr.py#L184"}}),wt=new J({props:{title:"DetrModel",local:"transformers.DetrModel",headingTag:"h2"}}),Dt=new w({props:{name:"class transformers.DetrModel",anchor:"transformers.DetrModel",parameters:[{name:"config",val:": DetrConfig"}],parametersDescription:[{anchor:"transformers.DetrModel.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrConfig">DetrConfig</a>) — | |
| Model configuration class with all the parameters of the model. Initializing with a config file does not | |
| load the weights associated with the model, only the configuration. Check out the | |
| <a href="/docs/transformers/pr_36839/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/modeling_detr.py#L1164"}}),xt=new w({props:{name:"forward",anchor:"transformers.DetrModel.forward",parameters:[{name:"pixel_values",val:": FloatTensor"},{name:"pixel_mask",val:": typing.Optional[torch.LongTensor] = None"},{name:"decoder_attention_mask",val:": typing.Optional[torch.FloatTensor] = None"},{name:"encoder_outputs",val:": typing.Optional[torch.FloatTensor] = None"},{name:"inputs_embeds",val:": typing.Optional[torch.FloatTensor] = None"},{name:"decoder_inputs_embeds",val:": typing.Optional[torch.FloatTensor] = None"},{name:"output_attentions",val:": typing.Optional[bool] = None"},{name:"output_hidden_states",val:": typing.Optional[bool] = None"},{name:"return_dict",val:": typing.Optional[bool] = None"}],parametersDescription:[{anchor:"transformers.DetrModel.forward.pixel_values",description:`<strong>pixel_values</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_channels, height, width)</code>) — | |
| Pixel values. Padding will be ignored by default should you provide it.</p> | |
| <p>Pixel values can be obtained using <a href="/docs/transformers/pr_36839/en/model_doc/auto#transformers.AutoImageProcessor">AutoImageProcessor</a>. See <a href="/docs/transformers/pr_36839/en/model_doc/vilt#transformers.ViltFeatureExtractor.__call__">DetrImageProcessor.<strong>call</strong>()</a> for details.`,name:"pixel_values"},{anchor:"transformers.DetrModel.forward.pixel_mask",description:`<strong>pixel_mask</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, height, width)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding pixel values. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for pixels that are real (i.e. <strong>not masked</strong>),</li> | |
| <li>0 for pixels that are padding (i.e. <strong>masked</strong>).</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"pixel_mask"},{anchor:"transformers.DetrModel.forward.decoder_attention_mask",description:`<strong>decoder_attention_mask</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries)</code>, <em>optional</em>) — | |
| Not used by default. Can be used to mask object queries.`,name:"decoder_attention_mask"},{anchor:"transformers.DetrModel.forward.encoder_outputs",description:`<strong>encoder_outputs</strong> (<code>tuple(tuple(torch.FloatTensor)</code>, <em>optional</em>) — | |
| Tuple consists of (<code>last_hidden_state</code>, <em>optional</em>: <code>hidden_states</code>, <em>optional</em>: <code>attentions</code>) | |
| <code>last_hidden_state</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) is a sequence of | |
| hidden-states at the output of the last layer of the encoder. Used in the cross-attention of the decoder.`,name:"encoder_outputs"},{anchor:"transformers.DetrModel.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing the flattened feature map (output of the backbone + projection layer), you | |
| can choose to directly pass a flattened representation of an image.`,name:"inputs_embeds"},{anchor:"transformers.DetrModel.forward.decoder_inputs_embeds",description:`<strong>decoder_inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of initializing the queries with a tensor of zeros, you can choose to directly pass an | |
| embedded representation.`,name:"decoder_inputs_embeds"},{anchor:"transformers.DetrModel.forward.output_attentions",description:`<strong>output_attentions</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the attentions tensors of all attention layers. See <code>attentions</code> under returned | |
| tensors for more detail.`,name:"output_attentions"},{anchor:"transformers.DetrModel.forward.output_hidden_states",description:`<strong>output_hidden_states</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the hidden states of all layers. See <code>hidden_states</code> under returned tensors for | |
| more detail.`,name:"output_hidden_states"},{anchor:"transformers.DetrModel.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return a <a href="/docs/transformers/pr_36839/en/main_classes/output#transformers.utils.ModelOutput">ModelOutput</a> instead of a plain tuple.`,name:"return_dict"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/modeling_detr.py#L1205",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.models.detr.modeling_detr.DetrModelOutput" | |
| >transformers.models.detr.modeling_detr.DetrModelOutput</a> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrConfig" | |
| >DetrConfig</a>) and inputs.</p> | |
| <ul> | |
| <li><strong>last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>) — Sequence of hidden-states at the output of the last layer of the decoder of the model.</li> | |
| <li><strong>decoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings + one for the output of each layer) of | |
| shape <code>(batch_size, sequence_length, hidden_size)</code>. Hidden-states of the decoder at the output of each | |
| layer plus the initial embedding outputs.</li> | |
| <li><strong>decoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the decoder, after the attention softmax, used to compute the | |
| weighted average in the self-attention heads.</li> | |
| <li><strong>cross_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the decoder’s cross-attention layer, after the attention softmax, | |
| used to compute the weighted average in the cross-attention heads.</li> | |
| <li><strong>encoder_last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — Sequence of hidden-states at the output of the last layer of the encoder of the model.</li> | |
| <li><strong>encoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings + one for the output of each layer) of | |
| shape <code>(batch_size, sequence_length, hidden_size)</code>. Hidden-states of the encoder at the output of each | |
| layer plus the initial embedding outputs.</li> | |
| <li><strong>encoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the encoder, after the attention softmax, used to compute the | |
| weighted average in the self-attention heads.</li> | |
| <li><strong>intermediate_hidden_states</strong> (<code>torch.FloatTensor</code> of shape <code>(config.decoder_layers, batch_size, sequence_length, hidden_size)</code>, <em>optional</em>, returned when <code>config.auxiliary_loss=True</code>) — Intermediate decoder activations, i.e. the output of each decoder layer, each of them gone through a | |
| layernorm.</li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.models.detr.modeling_detr.DetrModelOutput" | |
| >transformers.models.detr.modeling_detr.DetrModelOutput</a> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),me=new Es({props:{$$slots:{default:[Yr]},$$scope:{ctx:k}}}),pe=new In({props:{anchor:"transformers.DetrModel.forward.example",$$slots:{default:[Qr]},$$scope:{ctx:k}}}),Mt=new J({props:{title:"DetrForObjectDetection",local:"transformers.DetrForObjectDetection",headingTag:"h2"}}),Ft=new w({props:{name:"class transformers.DetrForObjectDetection",anchor:"transformers.DetrForObjectDetection",parameters:[{name:"config",val:": DetrConfig"}],parametersDescription:[{anchor:"transformers.DetrForObjectDetection.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrConfig">DetrConfig</a>) — | |
| Model configuration class with all the parameters of the model. Initializing with a config file does not | |
| load the weights associated with the model, only the configuration. Check out the | |
| <a href="/docs/transformers/pr_36839/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/modeling_detr.py#L1354"}}),jt=new w({props:{name:"forward",anchor:"transformers.DetrForObjectDetection.forward",parameters:[{name:"pixel_values",val:": FloatTensor"},{name:"pixel_mask",val:": typing.Optional[torch.LongTensor] = None"},{name:"decoder_attention_mask",val:": typing.Optional[torch.FloatTensor] = None"},{name:"encoder_outputs",val:": typing.Optional[torch.FloatTensor] = None"},{name:"inputs_embeds",val:": typing.Optional[torch.FloatTensor] = None"},{name:"decoder_inputs_embeds",val:": typing.Optional[torch.FloatTensor] = None"},{name:"labels",val:": typing.Optional[typing.List[dict]] = None"},{name:"output_attentions",val:": typing.Optional[bool] = None"},{name:"output_hidden_states",val:": typing.Optional[bool] = None"},{name:"return_dict",val:": typing.Optional[bool] = None"}],parametersDescription:[{anchor:"transformers.DetrForObjectDetection.forward.pixel_values",description:`<strong>pixel_values</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_channels, height, width)</code>) — | |
| Pixel values. Padding will be ignored by default should you provide it.</p> | |
| <p>Pixel values can be obtained using <a href="/docs/transformers/pr_36839/en/model_doc/auto#transformers.AutoImageProcessor">AutoImageProcessor</a>. See <a href="/docs/transformers/pr_36839/en/model_doc/vilt#transformers.ViltFeatureExtractor.__call__">DetrImageProcessor.<strong>call</strong>()</a> for details.`,name:"pixel_values"},{anchor:"transformers.DetrForObjectDetection.forward.pixel_mask",description:`<strong>pixel_mask</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, height, width)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding pixel values. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for pixels that are real (i.e. <strong>not masked</strong>),</li> | |
| <li>0 for pixels that are padding (i.e. <strong>masked</strong>).</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"pixel_mask"},{anchor:"transformers.DetrForObjectDetection.forward.decoder_attention_mask",description:`<strong>decoder_attention_mask</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries)</code>, <em>optional</em>) — | |
| Not used by default. Can be used to mask object queries.`,name:"decoder_attention_mask"},{anchor:"transformers.DetrForObjectDetection.forward.encoder_outputs",description:`<strong>encoder_outputs</strong> (<code>tuple(tuple(torch.FloatTensor)</code>, <em>optional</em>) — | |
| Tuple consists of (<code>last_hidden_state</code>, <em>optional</em>: <code>hidden_states</code>, <em>optional</em>: <code>attentions</code>) | |
| <code>last_hidden_state</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) is a sequence of | |
| hidden-states at the output of the last layer of the encoder. Used in the cross-attention of the decoder.`,name:"encoder_outputs"},{anchor:"transformers.DetrForObjectDetection.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing the flattened feature map (output of the backbone + projection layer), you | |
| can choose to directly pass a flattened representation of an image.`,name:"inputs_embeds"},{anchor:"transformers.DetrForObjectDetection.forward.decoder_inputs_embeds",description:`<strong>decoder_inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of initializing the queries with a tensor of zeros, you can choose to directly pass an | |
| embedded representation.`,name:"decoder_inputs_embeds"},{anchor:"transformers.DetrForObjectDetection.forward.output_attentions",description:`<strong>output_attentions</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the attentions tensors of all attention layers. See <code>attentions</code> under returned | |
| tensors for more detail.`,name:"output_attentions"},{anchor:"transformers.DetrForObjectDetection.forward.output_hidden_states",description:`<strong>output_hidden_states</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the hidden states of all layers. See <code>hidden_states</code> under returned tensors for | |
| more detail.`,name:"output_hidden_states"},{anchor:"transformers.DetrForObjectDetection.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return a <a href="/docs/transformers/pr_36839/en/main_classes/output#transformers.utils.ModelOutput">ModelOutput</a> instead of a plain tuple.`,name:"return_dict"},{anchor:"transformers.DetrForObjectDetection.forward.labels",description:`<strong>labels</strong> (<code>List[Dict]</code> of len <code>(batch_size,)</code>, <em>optional</em>) — | |
| Labels for computing the bipartite matching loss. List of dicts, each dictionary containing at least the | |
| following 2 keys: ‘class_labels’ and ‘boxes’ (the class labels and bounding boxes of an image in the batch | |
| respectively). The class labels themselves should be a <code>torch.LongTensor</code> of len <code>(number of bounding boxes in the image,)</code> and the boxes a <code>torch.FloatTensor</code> of shape <code>(number of bounding boxes in the image, 4)</code>.`,name:"labels"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/modeling_detr.py#L1379",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.models.detr.modeling_detr.DetrObjectDetectionOutput" | |
| >transformers.models.detr.modeling_detr.DetrObjectDetectionOutput</a> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrConfig" | |
| >DetrConfig</a>) and inputs.</p> | |
| <ul> | |
| <li><strong>loss</strong> (<code>torch.FloatTensor</code> of shape <code>(1,)</code>, <em>optional</em>, returned when <code>labels</code> are provided)) — Total loss as a linear combination of a negative log-likehood (cross-entropy) for class prediction and a | |
| bounding box loss. The latter is defined as a linear combination of the L1 loss and the generalized | |
| scale-invariant IoU loss.</li> | |
| <li><strong>loss_dict</strong> (<code>Dict</code>, <em>optional</em>) — A dictionary containing the individual losses. Useful for logging.</li> | |
| <li><strong>logits</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, num_classes + 1)</code>) — Classification logits (including no-object) for all queries.</li> | |
| <li><strong>pred_boxes</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, 4)</code>) — Normalized boxes coordinates for all queries, represented as (center_x, center_y, width, height). These | |
| values are normalized in [0, 1], relative to the size of each individual image in the batch (disregarding | |
| possible padding). You can use <a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor.post_process_object_detection" | |
| >post_process_object_detection()</a> to retrieve the | |
| unnormalized bounding boxes.</li> | |
| <li><strong>auxiliary_outputs</strong> (<code>list[Dict]</code>, <em>optional</em>) — Optional, only returned when auxilary losses are activated (i.e. <code>config.auxiliary_loss</code> is set to <code>True</code>) | |
| and labels are provided. It is a list of dictionaries containing the two above keys (<code>logits</code> and | |
| <code>pred_boxes</code>) for each decoder layer.</li> | |
| <li><strong>last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — Sequence of hidden-states at the output of the last layer of the decoder of the model.</li> | |
| <li><strong>decoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings + one for the output of each layer) of | |
| shape <code>(batch_size, sequence_length, hidden_size)</code>. Hidden-states of the decoder at the output of each | |
| layer plus the initial embedding outputs.</li> | |
| <li><strong>decoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the decoder, after the attention softmax, used to compute the | |
| weighted average in the self-attention heads.</li> | |
| <li><strong>cross_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the decoder’s cross-attention layer, after the attention softmax, | |
| used to compute the weighted average in the cross-attention heads.</li> | |
| <li><strong>encoder_last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — Sequence of hidden-states at the output of the last layer of the encoder of the model.</li> | |
| <li><strong>encoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings + one for the output of each layer) of | |
| shape <code>(batch_size, sequence_length, hidden_size)</code>. Hidden-states of the encoder at the output of each | |
| layer plus the initial embedding outputs.</li> | |
| <li><strong>encoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the encoder, after the attention softmax, used to compute the | |
| weighted average in the self-attention heads.</li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.models.detr.modeling_detr.DetrObjectDetectionOutput" | |
| >transformers.models.detr.modeling_detr.DetrObjectDetectionOutput</a> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),he=new Es({props:{$$slots:{default:[Kr]},$$scope:{ctx:k}}}),ge=new In({props:{anchor:"transformers.DetrForObjectDetection.forward.example",$$slots:{default:[ea]},$$scope:{ctx:k}}}),zt=new J({props:{title:"DetrForSegmentation",local:"transformers.DetrForSegmentation",headingTag:"h2"}}),$t=new w({props:{name:"class transformers.DetrForSegmentation",anchor:"transformers.DetrForSegmentation",parameters:[{name:"config",val:": DetrConfig"}],parametersDescription:[{anchor:"transformers.DetrForSegmentation.config",description:`<strong>config</strong> (<a href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrConfig">DetrConfig</a>) — | |
| Model configuration class with all the parameters of the model. Initializing with a config file does not | |
| load the weights associated with the model, only the configuration. Check out the | |
| <a href="/docs/transformers/pr_36839/en/main_classes/model#transformers.PreTrainedModel.from_pretrained">from_pretrained()</a> method to load the model weights.`,name:"config"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/modeling_detr.py#L1493"}}),kt=new w({props:{name:"forward",anchor:"transformers.DetrForSegmentation.forward",parameters:[{name:"pixel_values",val:": FloatTensor"},{name:"pixel_mask",val:": typing.Optional[torch.LongTensor] = None"},{name:"decoder_attention_mask",val:": typing.Optional[torch.FloatTensor] = None"},{name:"encoder_outputs",val:": typing.Optional[torch.FloatTensor] = None"},{name:"inputs_embeds",val:": typing.Optional[torch.FloatTensor] = None"},{name:"decoder_inputs_embeds",val:": typing.Optional[torch.FloatTensor] = None"},{name:"labels",val:": typing.Optional[typing.List[dict]] = None"},{name:"output_attentions",val:": typing.Optional[bool] = None"},{name:"output_hidden_states",val:": typing.Optional[bool] = None"},{name:"return_dict",val:": typing.Optional[bool] = None"}],parametersDescription:[{anchor:"transformers.DetrForSegmentation.forward.pixel_values",description:`<strong>pixel_values</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_channels, height, width)</code>) — | |
| Pixel values. Padding will be ignored by default should you provide it.</p> | |
| <p>Pixel values can be obtained using <a href="/docs/transformers/pr_36839/en/model_doc/auto#transformers.AutoImageProcessor">AutoImageProcessor</a>. See <a href="/docs/transformers/pr_36839/en/model_doc/vilt#transformers.ViltFeatureExtractor.__call__">DetrImageProcessor.<strong>call</strong>()</a> for details.`,name:"pixel_values"},{anchor:"transformers.DetrForSegmentation.forward.pixel_mask",description:`<strong>pixel_mask</strong> (<code>torch.LongTensor</code> of shape <code>(batch_size, height, width)</code>, <em>optional</em>) — | |
| Mask to avoid performing attention on padding pixel values. Mask values selected in <code>[0, 1]</code>:</p> | |
| <ul> | |
| <li>1 for pixels that are real (i.e. <strong>not masked</strong>),</li> | |
| <li>0 for pixels that are padding (i.e. <strong>masked</strong>).</li> | |
| </ul> | |
| <p><a href="../glossary#attention-mask">What are attention masks?</a>`,name:"pixel_mask"},{anchor:"transformers.DetrForSegmentation.forward.decoder_attention_mask",description:`<strong>decoder_attention_mask</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries)</code>, <em>optional</em>) — | |
| Not used by default. Can be used to mask object queries.`,name:"decoder_attention_mask"},{anchor:"transformers.DetrForSegmentation.forward.encoder_outputs",description:`<strong>encoder_outputs</strong> (<code>tuple(tuple(torch.FloatTensor)</code>, <em>optional</em>) — | |
| Tuple consists of (<code>last_hidden_state</code>, <em>optional</em>: <code>hidden_states</code>, <em>optional</em>: <code>attentions</code>) | |
| <code>last_hidden_state</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) is a sequence of | |
| hidden-states at the output of the last layer of the encoder. Used in the cross-attention of the decoder.`,name:"encoder_outputs"},{anchor:"transformers.DetrForSegmentation.forward.inputs_embeds",description:`<strong>inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of passing the flattened feature map (output of the backbone + projection layer), you | |
| can choose to directly pass a flattened representation of an image.`,name:"inputs_embeds"},{anchor:"transformers.DetrForSegmentation.forward.decoder_inputs_embeds",description:`<strong>decoder_inputs_embeds</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, hidden_size)</code>, <em>optional</em>) — | |
| Optionally, instead of initializing the queries with a tensor of zeros, you can choose to directly pass an | |
| embedded representation.`,name:"decoder_inputs_embeds"},{anchor:"transformers.DetrForSegmentation.forward.output_attentions",description:`<strong>output_attentions</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the attentions tensors of all attention layers. See <code>attentions</code> under returned | |
| tensors for more detail.`,name:"output_attentions"},{anchor:"transformers.DetrForSegmentation.forward.output_hidden_states",description:`<strong>output_hidden_states</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return the hidden states of all layers. See <code>hidden_states</code> under returned tensors for | |
| more detail.`,name:"output_hidden_states"},{anchor:"transformers.DetrForSegmentation.forward.return_dict",description:`<strong>return_dict</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to return a <a href="/docs/transformers/pr_36839/en/main_classes/output#transformers.utils.ModelOutput">ModelOutput</a> instead of a plain tuple.`,name:"return_dict"},{anchor:"transformers.DetrForSegmentation.forward.labels",description:`<strong>labels</strong> (<code>List[Dict]</code> of len <code>(batch_size,)</code>, <em>optional</em>) — | |
| Labels for computing the bipartite matching loss, DICE/F-1 loss and Focal loss. List of dicts, each | |
| dictionary containing at least the following 3 keys: ‘class_labels’, ‘boxes’ and ‘masks’ (the class labels, | |
| bounding boxes and segmentation masks of an image in the batch respectively). The class labels themselves | |
| should be a <code>torch.LongTensor</code> of len <code>(number of bounding boxes in the image,)</code>, the boxes a | |
| <code>torch.FloatTensor</code> of shape <code>(number of bounding boxes in the image, 4)</code> and the masks a | |
| <code>torch.FloatTensor</code> of shape <code>(number of bounding boxes in the image, height, width)</code>.`,name:"labels"}],source:"https://github.com/huggingface/transformers/blob/vr_36839/src/transformers/models/detr/modeling_detr.py#L1522",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A <a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.models.detr.modeling_detr.DetrSegmentationOutput" | |
| >transformers.models.detr.modeling_detr.DetrSegmentationOutput</a> or a tuple of | |
| <code>torch.FloatTensor</code> (if <code>return_dict=False</code> is passed or when <code>config.return_dict=False</code>) comprising various | |
| elements depending on the configuration (<a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrConfig" | |
| >DetrConfig</a>) and inputs.</p> | |
| <ul> | |
| <li><strong>loss</strong> (<code>torch.FloatTensor</code> of shape <code>(1,)</code>, <em>optional</em>, returned when <code>labels</code> are provided)) — Total loss as a linear combination of a negative log-likehood (cross-entropy) for class prediction and a | |
| bounding box loss. The latter is defined as a linear combination of the L1 loss and the generalized | |
| scale-invariant IoU loss.</li> | |
| <li><strong>loss_dict</strong> (<code>Dict</code>, <em>optional</em>) — A dictionary containing the individual losses. Useful for logging.</li> | |
| <li><strong>logits</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, num_classes + 1)</code>) — Classification logits (including no-object) for all queries.</li> | |
| <li><strong>pred_boxes</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, 4)</code>) — Normalized boxes coordinates for all queries, represented as (center_x, center_y, width, height). These | |
| values are normalized in [0, 1], relative to the size of each individual image in the batch (disregarding | |
| possible padding). You can use <a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor.post_process_object_detection" | |
| >post_process_object_detection()</a> to retrieve the | |
| unnormalized bounding boxes.</li> | |
| <li><strong>pred_masks</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, num_queries, height/4, width/4)</code>) — Segmentation masks logits for all queries. See also | |
| <a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor.post_process_semantic_segmentation" | |
| >post_process_semantic_segmentation()</a> or | |
| <a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor.post_process_instance_segmentation" | |
| >post_process_instance_segmentation()</a> | |
| <a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.DetrImageProcessor.post_process_panoptic_segmentation" | |
| >post_process_panoptic_segmentation()</a> to evaluate semantic, instance and panoptic | |
| segmentation masks respectively.</li> | |
| <li><strong>auxiliary_outputs</strong> (<code>list[Dict]</code>, <em>optional</em>) — Optional, only returned when auxiliary losses are activated (i.e. <code>config.auxiliary_loss</code> is set to <code>True</code>) | |
| and labels are provided. It is a list of dictionaries containing the two above keys (<code>logits</code> and | |
| <code>pred_boxes</code>) for each decoder layer.</li> | |
| <li><strong>last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — Sequence of hidden-states at the output of the last layer of the decoder of the model.</li> | |
| <li><strong>decoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings + one for the output of each layer) of | |
| shape <code>(batch_size, sequence_length, hidden_size)</code>. Hidden-states of the decoder at the output of each | |
| layer plus the initial embedding outputs.</li> | |
| <li><strong>decoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the decoder, after the attention softmax, used to compute the | |
| weighted average in the self-attention heads.</li> | |
| <li><strong>cross_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the decoder’s cross-attention layer, after the attention softmax, | |
| used to compute the weighted average in the cross-attention heads.</li> | |
| <li><strong>encoder_last_hidden_state</strong> (<code>torch.FloatTensor</code> of shape <code>(batch_size, sequence_length, hidden_size)</code>, <em>optional</em>) — Sequence of hidden-states at the output of the last layer of the encoder of the model.</li> | |
| <li><strong>encoder_hidden_states</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_hidden_states=True</code> is passed or when <code>config.output_hidden_states=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for the output of the embeddings + one for the output of each layer) of | |
| shape <code>(batch_size, sequence_length, hidden_size)</code>. Hidden-states of the encoder at the output of each | |
| layer plus the initial embedding outputs.</li> | |
| <li><strong>encoder_attentions</strong> (<code>tuple(torch.FloatTensor)</code>, <em>optional</em>, returned when <code>output_attentions=True</code> is passed or when <code>config.output_attentions=True</code>) — Tuple of <code>torch.FloatTensor</code> (one for each layer) of shape <code>(batch_size, num_heads, sequence_length, sequence_length)</code>. Attentions weights of the encoder, after the attention softmax, used to compute the | |
| weighted average in the self-attention heads.</li> | |
| </ul> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><a | |
| href="/docs/transformers/pr_36839/en/model_doc/detr#transformers.models.detr.modeling_detr.DetrSegmentationOutput" | |
| >transformers.models.detr.modeling_detr.DetrSegmentationOutput</a> or <code>tuple(torch.FloatTensor)</code></p> | |
| `}}),fe=new Es({props:{$$slots:{default:[ta]},$$scope:{ctx:k}}}),ue=new In({props:{anchor:"transformers.DetrForSegmentation.forward.example",$$slots:{default:[oa]},$$scope:{ctx:k}}}),Ct=new Vr({props:{source:"https://github.com/huggingface/transformers/blob/main/docs/source/en/model_doc/detr.md"}}),{c(){c=a("meta"),x=s(),b=a("p"),y=s(),p(D.$$.fragment),l=s(),M=a("div"),M.innerHTML=Rs,wo=s(),p(Te.$$.fragment),Do=s(),ve=a("p"),ve.innerHTML=Zs,xo=s(),we=a("p"),we.textContent=Ss,Mo=s(),De=a("p"),De.innerHTML=Ws,Fo=s(),xe=a("p"),xe.innerHTML=Hs,jo=s(),p(Me.$$.fragment),zo=s(),Fe=a("p"),Fe.innerHTML=Bs,$o=s(),je=a("p"),je.innerHTML=As,ko=s(),ze=a("p"),ze.innerHTML=Gs,Co=s(),$e=a("p"),$e.innerHTML=Vs,Io=s(),ke=a("p"),ke.innerHTML=Xs,Po=s(),p(Ce.$$.fragment),qo=s(),Ie=a("ul"),Ie.innerHTML=Ys,Oo=s(),Pe=a("p"),Pe.textContent=Qs,No=s(),qe=a("p"),qe.textContent=Ks,Jo=s(),p(Oe.$$.fragment),Lo=s(),Ne=a("p"),Ne.textContent=er,Uo=s(),p(Je.$$.fragment),Eo=s(),Le=a("p"),Le.textContent=tr,Ro=s(),p(Ue.$$.fragment),Zo=s(),Ee=a("p"),Ee.textContent=or,So=s(),Re=a("table"),Re.innerHTML=nr,Wo=s(),Ze=a("p"),Ze.innerHTML=sr,Ho=s(),p(Se.$$.fragment),Bo=s(),We=a("p"),We.textContent=rr,Ao=s(),p(He.$$.fragment),Go=s(),Be=a("ul"),Be.innerHTML=ar,Vo=s(),Ae=a("p"),Ae.textContent=ir,Xo=s(),p(Ge.$$.fragment),Yo=s(),C=a("div"),p(Ve.$$.fragment),Pn=s(),Ut=a("p"),Ut.innerHTML=dr,qn=s(),Et=a("p"),Et.innerHTML=cr,On=s(),p(G.$$.fragment),Nn=s(),V=a("div"),p(Xe.$$.fragment),Jn=s(),Rt=a("p"),Rt.innerHTML=lr,Qo=s(),p(Ye.$$.fragment),Ko=s(),F=a("div"),p(Qe.$$.fragment),Ln=s(),Zt=a("p"),Zt.textContent=mr,Un=s(),X=a("div"),p(Ke.$$.fragment),En=s(),St=a("p"),St.textContent=pr,Rn=s(),Y=a("div"),p(et.$$.fragment),Zn=s(),Wt=a("p"),Wt.innerHTML=hr,Sn=s(),Q=a("div"),p(tt.$$.fragment),Wn=s(),Ht=a("p"),Ht.innerHTML=gr,Hn=s(),K=a("div"),p(ot.$$.fragment),Bn=s(),Bt=a("p"),Bt.innerHTML=fr,An=s(),ee=a("div"),p(nt.$$.fragment),Gn=s(),At=a("p"),At.innerHTML=ur,en=s(),p(st.$$.fragment),tn=s(),j=a("div"),p(rt.$$.fragment),Vn=s(),Gt=a("p"),Gt.textContent=_r,Xn=s(),te=a("div"),p(at.$$.fragment),Yn=s(),Vt=a("p"),Vt.textContent=br,Qn=s(),oe=a("div"),p(it.$$.fragment),Kn=s(),Xt=a("p"),Xt.innerHTML=yr,es=s(),ne=a("div"),p(dt.$$.fragment),ts=s(),Yt=a("p"),Yt.innerHTML=Tr,os=s(),se=a("div"),p(ct.$$.fragment),ns=s(),Qt=a("p"),Qt.innerHTML=vr,ss=s(),re=a("div"),p(lt.$$.fragment),rs=s(),Kt=a("p"),Kt.innerHTML=wr,on=s(),p(mt.$$.fragment),nn=s(),z=a("div"),p(pt.$$.fragment),as=s(),ae=a("div"),p(ht.$$.fragment),is=s(),eo=a("p"),eo.textContent=Dr,ds=s(),ie=a("div"),p(gt.$$.fragment),cs=s(),to=a("p"),to.innerHTML=xr,ls=s(),de=a("div"),p(ft.$$.fragment),ms=s(),oo=a("p"),oo.innerHTML=Mr,ps=s(),ce=a("div"),p(ut.$$.fragment),hs=s(),no=a("p"),no.innerHTML=Fr,gs=s(),le=a("div"),p(_t.$$.fragment),fs=s(),so=a("p"),so.innerHTML=jr,sn=s(),p(bt.$$.fragment),rn=s(),H=a("div"),p(yt.$$.fragment),us=s(),ro=a("p"),ro.textContent=zr,an=s(),B=a("div"),p(Tt.$$.fragment),_s=s(),ao=a("p"),ao.innerHTML=$r,dn=s(),A=a("div"),p(vt.$$.fragment),bs=s(),io=a("p"),io.innerHTML=kr,cn=s(),p(wt.$$.fragment),ln=s(),I=a("div"),p(Dt.$$.fragment),ys=s(),co=a("p"),co.textContent=Cr,Ts=s(),lo=a("p"),lo.innerHTML=Ir,vs=s(),mo=a("p"),mo.innerHTML=Pr,ws=s(),L=a("div"),p(xt.$$.fragment),Ds=s(),po=a("p"),po.innerHTML=qr,xs=s(),p(me.$$.fragment),Ms=s(),p(pe.$$.fragment),mn=s(),p(Mt.$$.fragment),pn=s(),P=a("div"),p(Ft.$$.fragment),Fs=s(),ho=a("p"),ho.textContent=Or,js=s(),go=a("p"),go.innerHTML=Nr,zs=s(),fo=a("p"),fo.innerHTML=Jr,$s=s(),U=a("div"),p(jt.$$.fragment),ks=s(),uo=a("p"),uo.innerHTML=Lr,Cs=s(),p(he.$$.fragment),Is=s(),p(ge.$$.fragment),hn=s(),p(zt.$$.fragment),gn=s(),q=a("div"),p($t.$$.fragment),Ps=s(),_o=a("p"),_o.textContent=Ur,qs=s(),bo=a("p"),bo.innerHTML=Er,Os=s(),yo=a("p"),yo.innerHTML=Rr,Ns=s(),E=a("div"),p(kt.$$.fragment),Js=s(),To=a("p"),To.innerHTML=Zr,Ls=s(),p(fe.$$.fragment),Us=s(),p(ue.$$.fragment),fn=s(),p(Ct.$$.fragment),un=s(),vo=a("p"),this.h()},l(e){const o=Ar("svelte-u9bgzb",document.head);c=i(o,"META",{name:!0,content:!0}),o.forEach(t),x=r(e),b=i(e,"P",{}),v(b).forEach(t),y=r(e),h(D.$$.fragment,e),l=r(e),M=i(e,"DIV",{class:!0,"data-svelte-h":!0}),m(M)!=="svelte-13t8s2t"&&(M.innerHTML=Rs),wo=r(e),h(Te.$$.fragment,e),Do=r(e),ve=i(e,"P",{"data-svelte-h":!0}),m(ve)!=="svelte-1619prt"&&(ve.innerHTML=Zs),xo=r(e),we=i(e,"P",{"data-svelte-h":!0}),m(we)!=="svelte-vfdo9a"&&(we.textContent=Ss),Mo=r(e),De=i(e,"P",{"data-svelte-h":!0}),m(De)!=="svelte-s87elr"&&(De.innerHTML=Ws),Fo=r(e),xe=i(e,"P",{"data-svelte-h":!0}),m(xe)!=="svelte-1gw5wuz"&&(xe.innerHTML=Hs),jo=r(e),h(Me.$$.fragment,e),zo=r(e),Fe=i(e,"P",{"data-svelte-h":!0}),m(Fe)!=="svelte-ivsinj"&&(Fe.innerHTML=Bs),$o=r(e),je=i(e,"P",{"data-svelte-h":!0}),m(je)!=="svelte-x01cd0"&&(je.innerHTML=As),ko=r(e),ze=i(e,"P",{"data-svelte-h":!0}),m(ze)!=="svelte-i2avyi"&&(ze.innerHTML=Gs),Co=r(e),$e=i(e,"P",{"data-svelte-h":!0}),m($e)!=="svelte-351ubi"&&($e.innerHTML=Vs),Io=r(e),ke=i(e,"P",{"data-svelte-h":!0}),m(ke)!=="svelte-1evrr5u"&&(ke.innerHTML=Xs),Po=r(e),h(Ce.$$.fragment,e),qo=r(e),Ie=i(e,"UL",{"data-svelte-h":!0}),m(Ie)!=="svelte-1re8hph"&&(Ie.innerHTML=Ys),Oo=r(e),Pe=i(e,"P",{"data-svelte-h":!0}),m(Pe)!=="svelte-5tif3l"&&(Pe.textContent=Qs),No=r(e),qe=i(e,"P",{"data-svelte-h":!0}),m(qe)!=="svelte-ixg096"&&(qe.textContent=Ks),Jo=r(e),h(Oe.$$.fragment,e),Lo=r(e),Ne=i(e,"P",{"data-svelte-h":!0}),m(Ne)!=="svelte-14rhv4g"&&(Ne.textContent=er),Uo=r(e),h(Je.$$.fragment,e),Eo=r(e),Le=i(e,"P",{"data-svelte-h":!0}),m(Le)!=="svelte-1hfzzjq"&&(Le.textContent=tr),Ro=r(e),h(Ue.$$.fragment,e),Zo=r(e),Ee=i(e,"P",{"data-svelte-h":!0}),m(Ee)!=="svelte-1e3p89m"&&(Ee.textContent=or),So=r(e),Re=i(e,"TABLE",{"data-svelte-h":!0}),m(Re)!=="svelte-1xic86u"&&(Re.innerHTML=nr),Wo=r(e),Ze=i(e,"P",{"data-svelte-h":!0}),m(Ze)!=="svelte-bmlqwb"&&(Ze.innerHTML=sr),Ho=r(e),h(Se.$$.fragment,e),Bo=r(e),We=i(e,"P",{"data-svelte-h":!0}),m(We)!=="svelte-5jc6k6"&&(We.textContent=rr),Ao=r(e),h(He.$$.fragment,e),Go=r(e),Be=i(e,"UL",{"data-svelte-h":!0}),m(Be)!=="svelte-89455k"&&(Be.innerHTML=ar),Vo=r(e),Ae=i(e,"P",{"data-svelte-h":!0}),m(Ae)!=="svelte-1xesile"&&(Ae.textContent=ir),Xo=r(e),h(Ge.$$.fragment,e),Yo=r(e),C=i(e,"DIV",{class:!0});var N=v(C);h(Ve.$$.fragment,N),Pn=r(N),Ut=i(N,"P",{"data-svelte-h":!0}),m(Ut)!=="svelte-c1nrvn"&&(Ut.innerHTML=dr),qn=r(N),Et=i(N,"P",{"data-svelte-h":!0}),m(Et)!=="svelte-4potwj"&&(Et.innerHTML=cr),On=r(N),h(G.$$.fragment,N),Nn=r(N),V=i(N,"DIV",{class:!0});var It=v(V);h(Xe.$$.fragment,It),Jn=r(It),Rt=i(It,"P",{"data-svelte-h":!0}),m(Rt)!=="svelte-163fm8q"&&(Rt.innerHTML=lr),It.forEach(t),N.forEach(t),Qo=r(e),h(Ye.$$.fragment,e),Ko=r(e),F=i(e,"DIV",{class:!0});var $=v(F);h(Qe.$$.fragment,$),Ln=r($),Zt=i($,"P",{"data-svelte-h":!0}),m(Zt)!=="svelte-19j0nu1"&&(Zt.textContent=mr),Un=r($),X=i($,"DIV",{class:!0});var Pt=v(X);h(Ke.$$.fragment,Pt),En=r(Pt),St=i(Pt,"P",{"data-svelte-h":!0}),m(St)!=="svelte-jgz2ra"&&(St.textContent=pr),Pt.forEach(t),Rn=r($),Y=i($,"DIV",{class:!0});var qt=v(Y);h(et.$$.fragment,qt),Zn=r(qt),Wt=i(qt,"P",{"data-svelte-h":!0}),m(Wt)!=="svelte-17ugcuz"&&(Wt.innerHTML=hr),qt.forEach(t),Sn=r($),Q=i($,"DIV",{class:!0});var Ot=v(Q);h(tt.$$.fragment,Ot),Wn=r(Ot),Ht=i(Ot,"P",{"data-svelte-h":!0}),m(Ht)!=="svelte-xiyxbg"&&(Ht.innerHTML=gr),Ot.forEach(t),Hn=r($),K=i($,"DIV",{class:!0});var Nt=v(K);h(ot.$$.fragment,Nt),Bn=r(Nt),Bt=i(Nt,"P",{"data-svelte-h":!0}),m(Bt)!=="svelte-1tvucik"&&(Bt.innerHTML=fr),Nt.forEach(t),An=r($),ee=i($,"DIV",{class:!0});var bn=v(ee);h(nt.$$.fragment,bn),Gn=r(bn),At=i(bn,"P",{"data-svelte-h":!0}),m(At)!=="svelte-1e25nda"&&(At.innerHTML=ur),bn.forEach(t),$.forEach(t),en=r(e),h(st.$$.fragment,e),tn=r(e),j=i(e,"DIV",{class:!0});var O=v(j);h(rt.$$.fragment,O),Vn=r(O),Gt=i(O,"P",{"data-svelte-h":!0}),m(Gt)!=="svelte-1dqjg9j"&&(Gt.textContent=_r),Xn=r(O),te=i(O,"DIV",{class:!0});var yn=v(te);h(at.$$.fragment,yn),Yn=r(yn),Vt=i(yn,"P",{"data-svelte-h":!0}),m(Vt)!=="svelte-1x3yxsa"&&(Vt.textContent=br),yn.forEach(t),Qn=r(O),oe=i(O,"DIV",{class:!0});var Tn=v(oe);h(it.$$.fragment,Tn),Kn=r(Tn),Xt=i(Tn,"P",{"data-svelte-h":!0}),m(Xt)!=="svelte-17ugcuz"&&(Xt.innerHTML=yr),Tn.forEach(t),es=r(O),ne=i(O,"DIV",{class:!0});var vn=v(ne);h(dt.$$.fragment,vn),ts=r(vn),Yt=i(vn,"P",{"data-svelte-h":!0}),m(Yt)!=="svelte-xiyxbg"&&(Yt.innerHTML=Tr),vn.forEach(t),os=r(O),se=i(O,"DIV",{class:!0});var wn=v(se);h(ct.$$.fragment,wn),ns=r(wn),Qt=i(wn,"P",{"data-svelte-h":!0}),m(Qt)!=="svelte-1tvucik"&&(Qt.innerHTML=vr),wn.forEach(t),ss=r(O),re=i(O,"DIV",{class:!0});var Dn=v(re);h(lt.$$.fragment,Dn),rs=r(Dn),Kt=i(Dn,"P",{"data-svelte-h":!0}),m(Kt)!=="svelte-1e25nda"&&(Kt.innerHTML=wr),Dn.forEach(t),O.forEach(t),on=r(e),h(mt.$$.fragment,e),nn=r(e),z=i(e,"DIV",{class:!0});var R=v(z);h(pt.$$.fragment,R),as=r(R),ae=i(R,"DIV",{class:!0});var xn=v(ae);h(ht.$$.fragment,xn),is=r(xn),eo=i(xn,"P",{"data-svelte-h":!0}),m(eo)!=="svelte-khengj"&&(eo.textContent=Dr),xn.forEach(t),ds=r(R),ie=i(R,"DIV",{class:!0});var Mn=v(ie);h(gt.$$.fragment,Mn),cs=r(Mn),to=i(Mn,"P",{"data-svelte-h":!0}),m(to)!=="svelte-17ugcuz"&&(to.innerHTML=xr),Mn.forEach(t),ls=r(R),de=i(R,"DIV",{class:!0});var Fn=v(de);h(ft.$$.fragment,Fn),ms=r(Fn),oo=i(Fn,"P",{"data-svelte-h":!0}),m(oo)!=="svelte-xiyxbg"&&(oo.innerHTML=Mr),Fn.forEach(t),ps=r(R),ce=i(R,"DIV",{class:!0});var jn=v(ce);h(ut.$$.fragment,jn),hs=r(jn),no=i(jn,"P",{"data-svelte-h":!0}),m(no)!=="svelte-1tvucik"&&(no.innerHTML=Fr),jn.forEach(t),gs=r(R),le=i(R,"DIV",{class:!0});var zn=v(le);h(_t.$$.fragment,zn),fs=r(zn),so=i(zn,"P",{"data-svelte-h":!0}),m(so)!=="svelte-1e25nda"&&(so.innerHTML=jr),zn.forEach(t),R.forEach(t),sn=r(e),h(bt.$$.fragment,e),rn=r(e),H=i(e,"DIV",{class:!0});var $n=v(H);h(yt.$$.fragment,$n),us=r($n),ro=i($n,"P",{"data-svelte-h":!0}),m(ro)!=="svelte-1ya2yj5"&&(ro.textContent=zr),$n.forEach(t),an=r(e),B=i(e,"DIV",{class:!0});var kn=v(B);h(Tt.$$.fragment,kn),_s=r(kn),ao=i(kn,"P",{"data-svelte-h":!0}),m(ao)!=="svelte-1xd4o8p"&&(ao.innerHTML=$r),kn.forEach(t),dn=r(e),A=i(e,"DIV",{class:!0});var Cn=v(A);h(vt.$$.fragment,Cn),bs=r(Cn),io=i(Cn,"P",{"data-svelte-h":!0}),m(io)!=="svelte-1otgkvd"&&(io.innerHTML=kr),Cn.forEach(t),cn=r(e),h(wt.$$.fragment,e),ln=r(e),I=i(e,"DIV",{class:!0});var Z=v(I);h(Dt.$$.fragment,Z),ys=r(Z),co=i(Z,"P",{"data-svelte-h":!0}),m(co)!=="svelte-esnh0n"&&(co.textContent=Cr),Ts=r(Z),lo=i(Z,"P",{"data-svelte-h":!0}),m(lo)!=="svelte-1k92amr"&&(lo.innerHTML=Ir),vs=r(Z),mo=i(Z,"P",{"data-svelte-h":!0}),m(mo)!=="svelte-hswkmf"&&(mo.innerHTML=Pr),ws=r(Z),L=i(Z,"DIV",{class:!0});var _e=v(L);h(xt.$$.fragment,_e),Ds=r(_e),po=i(_e,"P",{"data-svelte-h":!0}),m(po)!=="svelte-7bh4s1"&&(po.innerHTML=qr),xs=r(_e),h(me.$$.fragment,_e),Ms=r(_e),h(pe.$$.fragment,_e),_e.forEach(t),Z.forEach(t),mn=r(e),h(Mt.$$.fragment,e),pn=r(e),P=i(e,"DIV",{class:!0});var S=v(P);h(Ft.$$.fragment,S),Fs=r(S),ho=i(S,"P",{"data-svelte-h":!0}),m(ho)!=="svelte-dw6bi4"&&(ho.textContent=Or),js=r(S),go=i(S,"P",{"data-svelte-h":!0}),m(go)!=="svelte-1k92amr"&&(go.innerHTML=Nr),zs=r(S),fo=i(S,"P",{"data-svelte-h":!0}),m(fo)!=="svelte-hswkmf"&&(fo.innerHTML=Jr),$s=r(S),U=i(S,"DIV",{class:!0});var be=v(U);h(jt.$$.fragment,be),ks=r(be),uo=i(be,"P",{"data-svelte-h":!0}),m(uo)!=="svelte-7ablkp"&&(uo.innerHTML=Lr),Cs=r(be),h(he.$$.fragment,be),Is=r(be),h(ge.$$.fragment,be),be.forEach(t),S.forEach(t),hn=r(e),h(zt.$$.fragment,e),gn=r(e),q=i(e,"DIV",{class:!0});var W=v(q);h($t.$$.fragment,W),Ps=r(W),_o=i(W,"P",{"data-svelte-h":!0}),m(_o)!=="svelte-1yivh9f"&&(_o.textContent=Ur),qs=r(W),bo=i(W,"P",{"data-svelte-h":!0}),m(bo)!=="svelte-1k92amr"&&(bo.innerHTML=Er),Os=r(W),yo=i(W,"P",{"data-svelte-h":!0}),m(yo)!=="svelte-hswkmf"&&(yo.innerHTML=Rr),Ns=r(W),E=i(W,"DIV",{class:!0});var ye=v(E);h(kt.$$.fragment,ye),Js=r(ye),To=i(ye,"P",{"data-svelte-h":!0}),m(To)!=="svelte-jiv1b5"&&(To.innerHTML=Zr),Ls=r(ye),h(fe.$$.fragment,ye),Us=r(ye),h(ue.$$.fragment,ye),ye.forEach(t),W.forEach(t),fn=r(e),h(Ct.$$.fragment,e),un=r(e),vo=i(e,"P",{}),v(vo).forEach(t),this.h()},h(){T(c,"name","hf:doc:metadata"),T(c,"content",sa),T(M,"class","flex flex-wrap space-x-1"),T(V,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(C,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(X,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(Y,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(Q,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(K,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(ee,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(F,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(te,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(oe,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(ne,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(se,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(re,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(j,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(ae,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(ie,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(de,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(ce,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(le,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(z,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(H,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(B,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(A,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(L,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(I,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(U,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(P,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(E,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),T(q,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8")},m(e,o){n(document.head,c),d(e,x,o),d(e,b,o),d(e,y,o),g(D,e,o),d(e,l,o),d(e,M,o),d(e,wo,o),g(Te,e,o),d(e,Do,o),d(e,ve,o),d(e,xo,o),d(e,we,o),d(e,Mo,o),d(e,De,o),d(e,Fo,o),d(e,xe,o),d(e,jo,o),g(Me,e,o),d(e,zo,o),d(e,Fe,o),d(e,$o,o),d(e,je,o),d(e,ko,o),d(e,ze,o),d(e,Co,o),d(e,$e,o),d(e,Io,o),d(e,ke,o),d(e,Po,o),g(Ce,e,o),d(e,qo,o),d(e,Ie,o),d(e,Oo,o),d(e,Pe,o),d(e,No,o),d(e,qe,o),d(e,Jo,o),g(Oe,e,o),d(e,Lo,o),d(e,Ne,o),d(e,Uo,o),g(Je,e,o),d(e,Eo,o),d(e,Le,o),d(e,Ro,o),g(Ue,e,o),d(e,Zo,o),d(e,Ee,o),d(e,So,o),d(e,Re,o),d(e,Wo,o),d(e,Ze,o),d(e,Ho,o),g(Se,e,o),d(e,Bo,o),d(e,We,o),d(e,Ao,o),g(He,e,o),d(e,Go,o),d(e,Be,o),d(e,Vo,o),d(e,Ae,o),d(e,Xo,o),g(Ge,e,o),d(e,Yo,o),d(e,C,o),g(Ve,C,null),n(C,Pn),n(C,Ut),n(C,qn),n(C,Et),n(C,On),g(G,C,null),n(C,Nn),n(C,V),g(Xe,V,null),n(V,Jn),n(V,Rt),d(e,Qo,o),g(Ye,e,o),d(e,Ko,o),d(e,F,o),g(Qe,F,null),n(F,Ln),n(F,Zt),n(F,Un),n(F,X),g(Ke,X,null),n(X,En),n(X,St),n(F,Rn),n(F,Y),g(et,Y,null),n(Y,Zn),n(Y,Wt),n(F,Sn),n(F,Q),g(tt,Q,null),n(Q,Wn),n(Q,Ht),n(F,Hn),n(F,K),g(ot,K,null),n(K,Bn),n(K,Bt),n(F,An),n(F,ee),g(nt,ee,null),n(ee,Gn),n(ee,At),d(e,en,o),g(st,e,o),d(e,tn,o),d(e,j,o),g(rt,j,null),n(j,Vn),n(j,Gt),n(j,Xn),n(j,te),g(at,te,null),n(te,Yn),n(te,Vt),n(j,Qn),n(j,oe),g(it,oe,null),n(oe,Kn),n(oe,Xt),n(j,es),n(j,ne),g(dt,ne,null),n(ne,ts),n(ne,Yt),n(j,os),n(j,se),g(ct,se,null),n(se,ns),n(se,Qt),n(j,ss),n(j,re),g(lt,re,null),n(re,rs),n(re,Kt),d(e,on,o),g(mt,e,o),d(e,nn,o),d(e,z,o),g(pt,z,null),n(z,as),n(z,ae),g(ht,ae,null),n(ae,is),n(ae,eo),n(z,ds),n(z,ie),g(gt,ie,null),n(ie,cs),n(ie,to),n(z,ls),n(z,de),g(ft,de,null),n(de,ms),n(de,oo),n(z,ps),n(z,ce),g(ut,ce,null),n(ce,hs),n(ce,no),n(z,gs),n(z,le),g(_t,le,null),n(le,fs),n(le,so),d(e,sn,o),g(bt,e,o),d(e,rn,o),d(e,H,o),g(yt,H,null),n(H,us),n(H,ro),d(e,an,o),d(e,B,o),g(Tt,B,null),n(B,_s),n(B,ao),d(e,dn,o),d(e,A,o),g(vt,A,null),n(A,bs),n(A,io),d(e,cn,o),g(wt,e,o),d(e,ln,o),d(e,I,o),g(Dt,I,null),n(I,ys),n(I,co),n(I,Ts),n(I,lo),n(I,vs),n(I,mo),n(I,ws),n(I,L),g(xt,L,null),n(L,Ds),n(L,po),n(L,xs),g(me,L,null),n(L,Ms),g(pe,L,null),d(e,mn,o),g(Mt,e,o),d(e,pn,o),d(e,P,o),g(Ft,P,null),n(P,Fs),n(P,ho),n(P,js),n(P,go),n(P,zs),n(P,fo),n(P,$s),n(P,U),g(jt,U,null),n(U,ks),n(U,uo),n(U,Cs),g(he,U,null),n(U,Is),g(ge,U,null),d(e,hn,o),g(zt,e,o),d(e,gn,o),d(e,q,o),g($t,q,null),n(q,Ps),n(q,_o),n(q,qs),n(q,bo),n(q,Os),n(q,yo),n(q,Ns),n(q,E),g(kt,E,null),n(E,Js),n(E,To),n(E,Ls),g(fe,E,null),n(E,Us),g(ue,E,null),d(e,fn,o),g(Ct,e,o),d(e,un,o),d(e,vo,o),_n=!0},p(e,[o]){const N={};o&2&&(N.$$scope={dirty:o,ctx:e}),G.$set(N);const It={};o&2&&(It.$$scope={dirty:o,ctx:e}),me.$set(It);const $={};o&2&&($.$$scope={dirty:o,ctx:e}),pe.$set($);const Pt={};o&2&&(Pt.$$scope={dirty:o,ctx:e}),he.$set(Pt);const qt={};o&2&&(qt.$$scope={dirty:o,ctx:e}),ge.$set(qt);const Ot={};o&2&&(Ot.$$scope={dirty:o,ctx:e}),fe.$set(Ot);const Nt={};o&2&&(Nt.$$scope={dirty:o,ctx:e}),ue.$set(Nt)},i(e){_n||(f(D.$$.fragment,e),f(Te.$$.fragment,e),f(Me.$$.fragment,e),f(Ce.$$.fragment,e),f(Oe.$$.fragment,e),f(Je.$$.fragment,e),f(Ue.$$.fragment,e),f(Se.$$.fragment,e),f(He.$$.fragment,e),f(Ge.$$.fragment,e),f(Ve.$$.fragment,e),f(G.$$.fragment,e),f(Xe.$$.fragment,e),f(Ye.$$.fragment,e),f(Qe.$$.fragment,e),f(Ke.$$.fragment,e),f(et.$$.fragment,e),f(tt.$$.fragment,e),f(ot.$$.fragment,e),f(nt.$$.fragment,e),f(st.$$.fragment,e),f(rt.$$.fragment,e),f(at.$$.fragment,e),f(it.$$.fragment,e),f(dt.$$.fragment,e),f(ct.$$.fragment,e),f(lt.$$.fragment,e),f(mt.$$.fragment,e),f(pt.$$.fragment,e),f(ht.$$.fragment,e),f(gt.$$.fragment,e),f(ft.$$.fragment,e),f(ut.$$.fragment,e),f(_t.$$.fragment,e),f(bt.$$.fragment,e),f(yt.$$.fragment,e),f(Tt.$$.fragment,e),f(vt.$$.fragment,e),f(wt.$$.fragment,e),f(Dt.$$.fragment,e),f(xt.$$.fragment,e),f(me.$$.fragment,e),f(pe.$$.fragment,e),f(Mt.$$.fragment,e),f(Ft.$$.fragment,e),f(jt.$$.fragment,e),f(he.$$.fragment,e),f(ge.$$.fragment,e),f(zt.$$.fragment,e),f($t.$$.fragment,e),f(kt.$$.fragment,e),f(fe.$$.fragment,e),f(ue.$$.fragment,e),f(Ct.$$.fragment,e),_n=!0)},o(e){u(D.$$.fragment,e),u(Te.$$.fragment,e),u(Me.$$.fragment,e),u(Ce.$$.fragment,e),u(Oe.$$.fragment,e),u(Je.$$.fragment,e),u(Ue.$$.fragment,e),u(Se.$$.fragment,e),u(He.$$.fragment,e),u(Ge.$$.fragment,e),u(Ve.$$.fragment,e),u(G.$$.fragment,e),u(Xe.$$.fragment,e),u(Ye.$$.fragment,e),u(Qe.$$.fragment,e),u(Ke.$$.fragment,e),u(et.$$.fragment,e),u(tt.$$.fragment,e),u(ot.$$.fragment,e),u(nt.$$.fragment,e),u(st.$$.fragment,e),u(rt.$$.fragment,e),u(at.$$.fragment,e),u(it.$$.fragment,e),u(dt.$$.fragment,e),u(ct.$$.fragment,e),u(lt.$$.fragment,e),u(mt.$$.fragment,e),u(pt.$$.fragment,e),u(ht.$$.fragment,e),u(gt.$$.fragment,e),u(ft.$$.fragment,e),u(ut.$$.fragment,e),u(_t.$$.fragment,e),u(bt.$$.fragment,e),u(yt.$$.fragment,e),u(Tt.$$.fragment,e),u(vt.$$.fragment,e),u(wt.$$.fragment,e),u(Dt.$$.fragment,e),u(xt.$$.fragment,e),u(me.$$.fragment,e),u(pe.$$.fragment,e),u(Mt.$$.fragment,e),u(Ft.$$.fragment,e),u(jt.$$.fragment,e),u(he.$$.fragment,e),u(ge.$$.fragment,e),u(zt.$$.fragment,e),u($t.$$.fragment,e),u(kt.$$.fragment,e),u(fe.$$.fragment,e),u(ue.$$.fragment,e),u(Ct.$$.fragment,e),_n=!1},d(e){e&&(t(x),t(b),t(y),t(l),t(M),t(wo),t(Do),t(ve),t(xo),t(we),t(Mo),t(De),t(Fo),t(xe),t(jo),t(zo),t(Fe),t($o),t(je),t(ko),t(ze),t(Co),t($e),t(Io),t(ke),t(Po),t(qo),t(Ie),t(Oo),t(Pe),t(No),t(qe),t(Jo),t(Lo),t(Ne),t(Uo),t(Eo),t(Le),t(Ro),t(Zo),t(Ee),t(So),t(Re),t(Wo),t(Ze),t(Ho),t(Bo),t(We),t(Ao),t(Go),t(Be),t(Vo),t(Ae),t(Xo),t(Yo),t(C),t(Qo),t(Ko),t(F),t(en),t(tn),t(j),t(on),t(nn),t(z),t(sn),t(rn),t(H),t(an),t(B),t(dn),t(A),t(cn),t(ln),t(I),t(mn),t(pn),t(P),t(hn),t(gn),t(q),t(fn),t(un),t(vo)),t(c),_(D,e),_(Te,e),_(Me,e),_(Ce,e),_(Oe,e),_(Je,e),_(Ue,e),_(Se,e),_(He,e),_(Ge,e),_(Ve),_(G),_(Xe),_(Ye,e),_(Qe),_(Ke),_(et),_(tt),_(ot),_(nt),_(st,e),_(rt),_(at),_(it),_(dt),_(ct),_(lt),_(mt,e),_(pt),_(ht),_(gt),_(ft),_(ut),_(_t),_(bt,e),_(yt),_(Tt),_(vt),_(wt,e),_(Dt),_(xt),_(me),_(pe),_(Mt,e),_(Ft),_(jt),_(he),_(ge),_(zt,e),_($t),_(kt),_(fe),_(ue),_(Ct,e)}}}const sa='{"title":"DETR","local":"detr","sections":[{"title":"Overview","local":"overview","sections":[],"depth":2},{"title":"How DETR works","local":"how-detr-works","sections":[],"depth":2},{"title":"Usage tips","local":"usage-tips","sections":[],"depth":2},{"title":"Resources","local":"resources","sections":[],"depth":2},{"title":"DetrConfig","local":"transformers.DetrConfig","sections":[],"depth":2},{"title":"DetrImageProcessor","local":"transformers.DetrImageProcessor","sections":[],"depth":2},{"title":"DetrImageProcessorFast","local":"transformers.DetrImageProcessorFast","sections":[],"depth":2},{"title":"DetrFeatureExtractor","local":"transformers.DetrFeatureExtractor","sections":[],"depth":2},{"title":"DETR specific outputs","local":"transformers.models.detr.modeling_detr.DetrModelOutput","sections":[],"depth":2},{"title":"DetrModel","local":"transformers.DetrModel","sections":[],"depth":2},{"title":"DetrForObjectDetection","local":"transformers.DetrForObjectDetection","sections":[],"depth":2},{"title":"DetrForSegmentation","local":"transformers.DetrForSegmentation","sections":[],"depth":2}],"depth":1}';function ra(k){return Wr(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class ga extends Hr{constructor(c){super(),Br(this,c,ra,na,Sr,{})}}export{ga as component}; | |
Xet Storage Details
- Size:
- 209 kB
- Xet hash:
- c40cd6db5787970260150489943c74f150aa17851b1d0b372db9f8d233f7ae8b
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.