Buckets:

download
raw
8.07 kB
import"../chunks/DsnmJJEf.js";import{i as w,h as J,C as T,H as t,a as g,E as j,s as v}from"../chunks/CD2rhSaz.js";import{p as U,o as R,s as e,f as V,a as n,b as x,d as o,n as W}from"../chunks/DmjbnfDo.js";import{T as k}from"../chunks/B2suExpn.js";const _='{"title":"DePlot","local":"deplot","sections":[{"title":"Overview","local":"overview","sections":[],"depth":2},{"title":"Usage example","local":"usage-example","sections":[],"depth":2},{"title":"Fine-tuning","local":"fine-tuning","sections":[],"depth":2}],"depth":1}';var G=o('<meta name="hf:doc:metadata"/>'),Z=o('<p>DePlot is a model trained using <code>Pix2Struct</code> architecture. For API reference, see <a href="pix2struct"><code>Pix2Struct</code> documentation</a>.</p>'),F=o(`<p></p> <p><em>This model was published in HF papers on 2022-12-20 and contributed to Hugging Face Transformers on 2023-06-20.</em></p> <!> <!> <!> <p>DePlot was proposed in the paper <a href="https://huggingface.co/papers/2212.10505" rel="nofollow">DePlot: One-shot visual language reasoning by plot-to-table translation</a> from Fangyu Liu, Julian Martin Eisenschlos, Francesco Piccinno, Syrine Krichene, Chenxi Pang, Kenton Lee, Mandar Joshi, Wenhu Chen, Nigel Collier, Yasemin Altun.</p> <p>The abstract of the paper states the following:</p> <p><em>Visual language such as charts and plots is ubiquitous in the human world. Comprehending plots and charts requires strong reasoning skills. Prior state-of-the-art (SOTA) models require at least tens of thousands of training examples and their reasoning capabilities are still much limited, especially on complex human-written queries. This paper presents the first one-shot solution to visual language reasoning. We decompose the challenge of visual language reasoning into two steps: (1) plot-to-text translation, and (2) reasoning over the translated text. The key in this method is a modality conversion module, named as DePlot, which translates the image of a plot or chart to a linearized table. The output of DePlot can then be directly used to prompt a pretrained large language model (LLM), exploiting the few-shot reasoning capabilities of LLMs. To obtain DePlot, we standardize the plot-to-table task by establishing unified task formats and metrics, and train DePlot end-to-end on this task. DePlot can then be used off-the-shelf together with LLMs in a plug-and-play fashion. Compared with a SOTA model finetuned on more than >28k data points, DePlot+LLM with just one-shot prompting achieves a 24.0% improvement over finetuned SOTA on human-written queries from the task of chart QA.</em></p> <p>DePlot is a model that is trained using <code>Pix2Struct</code> architecture. You can find more information about <code>Pix2Struct</code> in the <a href="https://huggingface.co/docs/transformers/main/en/model_doc/pix2struct" rel="nofollow">Pix2Struct documentation</a>.
DePlot is a Visual Question Answering subset of <code>Pix2Struct</code> architecture. It renders the input question on the image and predicts the answer.</p> <!> <p>Currently one checkpoint is available for DePlot:</p> <ul><li><code>google/deplot</code>: DePlot fine-tuned on ChartQA dataset</li></ul> <!> <!> <p>To fine-tune DePlot, refer to the pix2struct <a href="https://github.com/huggingface/notebooks/blob/main/examples/image_captioning_pix2struct.ipynb" rel="nofollow">fine-tuning notebook</a>. For <code>Pix2Struct</code> models, we have found out that fine-tuning the model with Adafactor and cosine learning rate scheduler leads to faster convergence:</p> <!> <!> <!> <p></p>`,1);function z(b,M){U(M,!1),R(()=>{new URLSearchParams(window.location.search).get("fw")}),w();var l=F();J("rqdnse",a=>{var s=G();v(s,"content",_),n(a,s)});var i=e(V(l),4);T(i,{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"});var r=e(i,2);t(r,{title:"DePlot",local:"deplot",headingTag:"h1"});var p=e(r,2);t(p,{title:"Overview",local:"overview",headingTag:"h2"});var c=e(p,10);t(c,{title:"Usage example",local:"usage-example",headingTag:"h2"});var d=e(c,6);g(d,{code:"aW1wb3J0JTIwcmVxdWVzdHMlMEFmcm9tJTIwUElMJTIwaW1wb3J0JTIwSW1hZ2UlMEElMEFmcm9tJTIwdHJhbnNmb3JtZXJzJTIwaW1wb3J0JTIwQXV0b1Byb2Nlc3NvciUyQyUyMFBpeDJTdHJ1Y3RGb3JDb25kaXRpb25hbEdlbmVyYXRpb24lMEElMEElMEFtb2RlbCUyMCUzRCUyMFBpeDJTdHJ1Y3RGb3JDb25kaXRpb25hbEdlbmVyYXRpb24uZnJvbV9wcmV0cmFpbmVkKCUyMmdvb2dsZSUyRmRlcGxvdCUyMiUyQyUyMGRldmljZV9tYXAlM0QlMjJhdXRvJTIyKSUwQXByb2Nlc3NvciUyMCUzRCUyMEF1dG9Qcm9jZXNzb3IuZnJvbV9wcmV0cmFpbmVkKCUyMmdvb2dsZSUyRmRlcGxvdCUyMiklMEF1cmwlMjAlM0QlMjAlMjJodHRwcyUzQSUyRiUyRnJhdy5naXRodWJ1c2VyY29udGVudC5jb20lMkZ2aXMtbmxwJTJGQ2hhcnRRQSUyRm1haW4lMkZDaGFydFFBJTI1MjBEYXRhc2V0JTJGdmFsJTJGcG5nJTJGNTA5MC5wbmclMjIlMEFpbWFnZSUyMCUzRCUyMEltYWdlLm9wZW4ocmVxdWVzdHMuZ2V0KHVybCUyQyUyMHN0cmVhbSUzRFRydWUpLnJhdyklMEElMEFpbnB1dHMlMjAlM0QlMjBwcm9jZXNzb3IoaW1hZ2VzJTNEaW1hZ2UlMkMlMjB0ZXh0JTNEJTIyR2VuZXJhdGUlMjB1bmRlcmx5aW5nJTIwZGF0YSUyMHRhYmxlJTIwb2YlMjB0aGUlMjBmaWd1cmUlMjBiZWxvdyUzQSUyMiUyQyUyMHJldHVybl90ZW5zb3JzJTNEJTIycHQlMjIpLnRvKG1vZGVsLmRldmljZSklMEFwcmVkaWN0aW9ucyUyMCUzRCUyMG1vZGVsLmdlbmVyYXRlKCoqaW5wdXRzJTJDJTIwbWF4X25ld190b2tlbnMlM0Q1MTIpJTBBcHJpbnQocHJvY2Vzc29yLmRlY29kZShwcmVkaWN0aW9ucyU1QjAlNUQlMkMlMjBza2lwX3NwZWNpYWxfdG9rZW5zJTNEVHJ1ZSkp",highlighted:`<span class="hljs-keyword">import</span> requests
<span class="hljs-keyword">from</span> PIL <span class="hljs-keyword">import</span> Image
<span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoProcessor, Pix2StructForConditionalGeneration
model = Pix2StructForConditionalGeneration.from_pretrained(<span class="hljs-string">&quot;google/deplot&quot;</span>, device_map=<span class="hljs-string">&quot;auto&quot;</span>)
processor = AutoProcessor.from_pretrained(<span class="hljs-string">&quot;google/deplot&quot;</span>)
url = <span class="hljs-string">&quot;https://raw.githubusercontent.com/vis-nlp/ChartQA/main/ChartQA%20Dataset/val/png/5090.png&quot;</span>
image = Image.<span class="hljs-built_in">open</span>(requests.get(url, stream=<span class="hljs-literal">True</span>).raw)
inputs = processor(images=image, text=<span class="hljs-string">&quot;Generate underlying data table of the figure below:&quot;</span>, return_tensors=<span class="hljs-string">&quot;pt&quot;</span>).to(model.device)
predictions = model.generate(**inputs, max_new_tokens=<span class="hljs-number">512</span>)
<span class="hljs-built_in">print</span>(processor.decode(predictions[<span class="hljs-number">0</span>], skip_special_tokens=<span class="hljs-literal">True</span>))`,lang:"python",wrap:!1});var h=e(d,2);t(h,{title:"Fine-tuning",local:"fine-tuning",headingTag:"h2"});var m=e(h,4);g(m,{code:"ZnJvbSUyMHRyYW5zZm9ybWVycy5vcHRpbWl6YXRpb24lMjBpbXBvcnQlMjBBZGFmYWN0b3IlMkMlMjBnZXRfY29zaW5lX3NjaGVkdWxlX3dpdGhfd2FybXVwJTBBJTBBJTBBb3B0aW1pemVyJTIwJTNEJTIwQWRhZmFjdG9yKHNlbGYucGFyYW1ldGVycygpJTJDJTIwc2NhbGVfcGFyYW1ldGVyJTNERmFsc2UlMkMlMjByZWxhdGl2ZV9zdGVwJTNERmFsc2UlMkMlMjBsciUzRDAuMDElMkMlMjB3ZWlnaHRfZGVjYXklM0QxZS0wNSklMEFzY2hlZHVsZXIlMjAlM0QlMjBnZXRfY29zaW5lX3NjaGVkdWxlX3dpdGhfd2FybXVwKG9wdGltaXplciUyQyUyMG51bV93YXJtdXBfc3RlcHMlM0QxMDAwJTJDJTIwbnVtX3RyYWluaW5nX3N0ZXBzJTNENDAwMDAp",highlighted:`<span class="hljs-keyword">from</span> transformers.optimization <span class="hljs-keyword">import</span> Adafactor, get_cosine_schedule_with_warmup
optimizer = Adafactor(<span class="hljs-variable language_">self</span>.parameters(), scale_parameter=<span class="hljs-literal">False</span>, relative_step=<span class="hljs-literal">False</span>, lr=<span class="hljs-number">0.01</span>, weight_decay=<span class="hljs-number">1e-05</span>)
scheduler = get_cosine_schedule_with_warmup(optimizer, num_warmup_steps=<span class="hljs-number">1000</span>, num_training_steps=<span class="hljs-number">40000</span>)`,lang:"python",wrap:!1});var u=e(m,2);k(u,{children:(a,s)=>{var f=Z();n(a,f)},$$slots:{default:!0}});var y=e(u,2);j(y,{source:"https://github.com/huggingface/transformers/blob/main/docs/source/en/model_doc/deplot.md"}),W(2),n(b,l),x()}export{z as component};

Xet Storage Details

Size:
8.07 kB
·
Xet hash:
7f54ed2f390c1ff35cbade7e1e3dcada192f9e64d1ae0a5ba4678d3f1db99daa

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.