Buckets:
| import{s as hm,o as fm,n as z}from"../chunks/scheduler.bdbef820.js";import{S as _m,i as bm,g as r,s as o,r as p,A as vm,h as a,f as d,c as n,j as T,u,x as l,k as w,y as e,a as $,v as g,d as h,t as f,w as _}from"../chunks/index.33f81d56.js";import{T as Eo}from"../chunks/Tip.34194030.js";import{D as x}from"../chunks/Docstring.64554317.js";import{C as ve}from"../chunks/CodeBlock.362b34a4.js";import{E as be}from"../chunks/ExampleCodeBlock.4f2252c6.js";import{H as ua,E as ym}from"../chunks/EditOnGithub.a9246e21.js";function Tm(C){let i,k='<a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a> 클래스는 🤗 Transformers 모델에 최적화되어 있으며, 다른 모델과 함께 사용될 때 예상치 못한 동작을 하게 될 수 있습니다. 자신만의 모델을 사용할 때는 다음을 확인하세요:',m,c,q='<li>모델은 항상 튜플이나 <a href="/docs/transformers/pr_34009/ko/main_classes/output#transformers.utils.ModelOutput">ModelOutput</a>의 서브클래스를 반환해야 합니다.</li> <li>모델은 <code>labels</code> 인자가 제공되면 손실을 계산할 수 있고, 모델이 튜플을 반환하는 경우 그 손실이 튜플의 첫 번째 요소로 반환되어야 합니다.</li> <li>모델은 여러 개의 레이블 인자를 수용할 수 있어야 하며, <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>에게 이름을 알리기 위해 <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.TrainingArguments">TrainingArguments</a>에서 <code>label_names</code>를 사용하지만, 그 중 어느 것도 <code>"label"</code>로 명명되어서는 안 됩니다.</li>';return{c(){i=r("p"),i.innerHTML=k,m=o(),c=r("ul"),c.innerHTML=q},l(s){i=a(s,"P",{"data-svelte-h":!0}),l(i)!=="svelte-17jaoct"&&(i.innerHTML=k),m=n(s),c=a(s,"UL",{"data-svelte-h":!0}),l(c)!=="svelte-1dykbii"&&(c.innerHTML=q)},m(s,A){$(s,i,A),$(s,m,A),$(s,c,A)},p:z,d(s){s&&(d(i),d(m),d(c))}}}function wm(C){let i,k=`To use this method, you need to have provided a <code>model_init</code> when initializing your <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>: we need to | |
| reinitialize the model at each new run. This is incompatible with the <code>optimizers</code> argument, so you need to | |
| subclass <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a> and override the method <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer.create_optimizer_and_scheduler">create_optimizer_and_scheduler()</a> for custom | |
| optimizer/scheduler.`;return{c(){i=r("p"),i.innerHTML=k},l(m){i=a(m,"P",{"data-svelte-h":!0}),l(i)!=="svelte-16l6z1o"&&(i.innerHTML=k)},m(m,c){$(m,i,c)},p:z,d(m){m&&d(i)}}}function xm(C){let i,k="Now when this method is run, you will see a report that will include: :",m,c,q;return c=new ve({props:{code:"aW5pdF9tZW1fY3B1X2FsbG9jX2RlbHRhJTIwJTIwJTIwJTNEJTIwJTIwJTIwJTIwJTIwMTMwMU1CJTBBaW5pdF9tZW1fY3B1X3BlYWtlZF9kZWx0YSUyMCUyMCUzRCUyMCUyMCUyMCUyMCUyMCUyMDE1NE1CJTBBaW5pdF9tZW1fZ3B1X2FsbG9jX2RlbHRhJTIwJTIwJTIwJTNEJTIwJTIwJTIwJTIwJTIwJTIwMjMwTUIlMEFpbml0X21lbV9ncHVfcGVha2VkX2RlbHRhJTIwJTIwJTNEJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwME1CJTBBdHJhaW5fbWVtX2NwdV9hbGxvY19kZWx0YSUyMCUyMCUzRCUyMCUyMCUyMCUyMCUyMDEzNDVNQiUwQXRyYWluX21lbV9jcHVfcGVha2VkX2RlbHRhJTIwJTNEJTIwJTIwJTIwJTIwJTIwJTIwJTIwJTIwME1CJTBBdHJhaW5fbWVtX2dwdV9hbGxvY19kZWx0YSUyMCUyMCUzRCUyMCUyMCUyMCUyMCUyMCUyMDY5M01CJTBBdHJhaW5fbWVtX2dwdV9wZWFrZWRfZGVsdGElMjAlM0QlMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjA3TUI=",highlighted:`<span class="hljs-attr">init_mem_cpu_alloc_delta</span> = <span class="hljs-number">1301</span>MB | |
| <span class="hljs-attr">init_mem_cpu_peaked_delta</span> = <span class="hljs-number">154</span>MB | |
| <span class="hljs-attr">init_mem_gpu_alloc_delta</span> = <span class="hljs-number">230</span>MB | |
| <span class="hljs-attr">init_mem_gpu_peaked_delta</span> = <span class="hljs-number">0</span>MB | |
| <span class="hljs-attr">train_mem_cpu_alloc_delta</span> = <span class="hljs-number">1345</span>MB | |
| <span class="hljs-attr">train_mem_cpu_peaked_delta</span> = <span class="hljs-number">0</span>MB | |
| <span class="hljs-attr">train_mem_gpu_alloc_delta</span> = <span class="hljs-number">693</span>MB | |
| <span class="hljs-attr">train_mem_gpu_peaked_delta</span> = <span class="hljs-number">7</span>MB`,wrap:!1}}),{c(){i=r("p"),i.textContent=k,m=o(),p(c.$$.fragment)},l(s){i=a(s,"P",{"data-svelte-h":!0}),l(i)!=="svelte-2hs27k"&&(i.textContent=k),m=n(s),u(c.$$.fragment,s)},m(s,A){$(s,i,A),$(s,m,A),g(c,s,A),q=!0},p:z,i(s){q||(h(c.$$.fragment,s),q=!0)},o(s){f(c.$$.fragment,s),q=!1},d(s){s&&(d(i),d(m)),_(c,s)}}}function qm(C){let i,k=`If your predictions or labels have different sequence length (for instance because you’re doing dynamic padding | |
| in a token classification task) the predictions will be padded (on the right) to allow for concatenation into | |
| one array. The padding index is -100.`;return{c(){i=r("p"),i.textContent=k},l(m){i=a(m,"P",{"data-svelte-h":!0}),l(i)!=="svelte-vq7haq"&&(i.textContent=k)},m(m,c){$(m,i,c)},p:z,d(m){m&&d(i)}}}function km(C){let i,k=`If your predictions or labels have different sequence lengths (for instance because you’re doing dynamic | |
| padding in a token classification task) the predictions will be padded (on the right) to allow for | |
| concatenation into one array. The padding index is -100.`;return{c(){i=r("p"),i.textContent=k},l(m){i=a(m,"P",{"data-svelte-h":!0}),l(i)!=="svelte-1c3ugfb"&&(i.textContent=k)},m(m,c){$(m,i,c)},p:z,d(m){m&&d(i)}}}function $m(C){let i,k="Example:",m,c,q;return c=new ve({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMFRyYWluaW5nQXJndW1lbnRzJTBBJTBBYXJncyUyMCUzRCUyMFRyYWluaW5nQXJndW1lbnRzKCUyMndvcmtpbmdfZGlyJTIyKSUwQWFyZ3MlMjAlM0QlMjBhcmdzLnNldF9kYXRhbG9hZGVyKHRyYWluX2JhdGNoX3NpemUlM0QxNiUyQyUyMGV2YWxfYmF0Y2hfc2l6ZSUzRDY0KSUwQWFyZ3MucGVyX2RldmljZV90cmFpbl9iYXRjaF9zaXpl",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> TrainingArguments | |
| <span class="hljs-meta">>>> </span>args = TrainingArguments(<span class="hljs-string">"working_dir"</span>) | |
| <span class="hljs-meta">>>> </span>args = args.set_dataloader(train_batch_size=<span class="hljs-number">16</span>, eval_batch_size=<span class="hljs-number">64</span>) | |
| <span class="hljs-meta">>>> </span>args.per_device_train_batch_size | |
| <span class="hljs-number">16</span>`,wrap:!1}}),{c(){i=r("p"),i.textContent=k,m=o(),p(c.$$.fragment)},l(s){i=a(s,"P",{"data-svelte-h":!0}),l(i)!=="svelte-11lpom8"&&(i.textContent=k),m=n(s),u(c.$$.fragment,s)},m(s,A){$(s,i,A),$(s,m,A),g(c,s,A),q=!0},p:z,i(s){q||(h(c.$$.fragment,s),q=!0)},o(s){f(c.$$.fragment,s),q=!1},d(s){s&&(d(i),d(m)),_(c,s)}}}function Am(C){let i,k="Example:",m,c,q;return c=new ve({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMFRyYWluaW5nQXJndW1lbnRzJTBBJTBBYXJncyUyMCUzRCUyMFRyYWluaW5nQXJndW1lbnRzKCUyMndvcmtpbmdfZGlyJTIyKSUwQWFyZ3MlMjAlM0QlMjBhcmdzLnNldF9ldmFsdWF0ZShzdHJhdGVneSUzRCUyMnN0ZXBzJTIyJTJDJTIwc3RlcHMlM0QxMDApJTBBYXJncy5ldmFsX3N0ZXBz",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> TrainingArguments | |
| <span class="hljs-meta">>>> </span>args = TrainingArguments(<span class="hljs-string">"working_dir"</span>) | |
| <span class="hljs-meta">>>> </span>args = args.set_evaluate(strategy=<span class="hljs-string">"steps"</span>, steps=<span class="hljs-number">100</span>) | |
| <span class="hljs-meta">>>> </span>args.eval_steps | |
| <span class="hljs-number">100</span>`,wrap:!1}}),{c(){i=r("p"),i.textContent=k,m=o(),p(c.$$.fragment)},l(s){i=a(s,"P",{"data-svelte-h":!0}),l(i)!=="svelte-11lpom8"&&(i.textContent=k),m=n(s),u(c.$$.fragment,s)},m(s,A){$(s,i,A),$(s,m,A),g(c,s,A),q=!0},p:z,i(s){q||(h(c.$$.fragment,s),q=!0)},o(s){f(c.$$.fragment,s),q=!1},d(s){s&&(d(i),d(m)),_(c,s)}}}function Sm(C){let i,k="Example:",m,c,q;return c=new ve({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMFRyYWluaW5nQXJndW1lbnRzJTBBJTBBYXJncyUyMCUzRCUyMFRyYWluaW5nQXJndW1lbnRzKCUyMndvcmtpbmdfZGlyJTIyKSUwQWFyZ3MlMjAlM0QlMjBhcmdzLnNldF9sb2dnaW5nKHN0cmF0ZWd5JTNEJTIyc3RlcHMlMjIlMkMlMjBzdGVwcyUzRDEwMCklMEFhcmdzLmxvZ2dpbmdfc3RlcHM=",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> TrainingArguments | |
| <span class="hljs-meta">>>> </span>args = TrainingArguments(<span class="hljs-string">"working_dir"</span>) | |
| <span class="hljs-meta">>>> </span>args = args.set_logging(strategy=<span class="hljs-string">"steps"</span>, steps=<span class="hljs-number">100</span>) | |
| <span class="hljs-meta">>>> </span>args.logging_steps | |
| <span class="hljs-number">100</span>`,wrap:!1}}),{c(){i=r("p"),i.textContent=k,m=o(),p(c.$$.fragment)},l(s){i=a(s,"P",{"data-svelte-h":!0}),l(i)!=="svelte-11lpom8"&&(i.textContent=k),m=n(s),u(c.$$.fragment,s)},m(s,A){$(s,i,A),$(s,m,A),g(c,s,A),q=!0},p:z,i(s){q||(h(c.$$.fragment,s),q=!0)},o(s){f(c.$$.fragment,s),q=!1},d(s){s&&(d(i),d(m)),_(c,s)}}}function Cm(C){let i,k="Example:",m,c,q;return c=new ve({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMFRyYWluaW5nQXJndW1lbnRzJTBBJTBBYXJncyUyMCUzRCUyMFRyYWluaW5nQXJndW1lbnRzKCUyMndvcmtpbmdfZGlyJTIyKSUwQWFyZ3MlMjAlM0QlMjBhcmdzLnNldF9scl9zY2hlZHVsZXIobmFtZSUzRCUyMmNvc2luZSUyMiUyQyUyMHdhcm11cF9yYXRpbyUzRDAuMDUpJTBBYXJncy53YXJtdXBfcmF0aW8=",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> TrainingArguments | |
| <span class="hljs-meta">>>> </span>args = TrainingArguments(<span class="hljs-string">"working_dir"</span>) | |
| <span class="hljs-meta">>>> </span>args = args.set_lr_scheduler(name=<span class="hljs-string">"cosine"</span>, warmup_ratio=<span class="hljs-number">0.05</span>) | |
| <span class="hljs-meta">>>> </span>args.warmup_ratio | |
| <span class="hljs-number">0.05</span>`,wrap:!1}}),{c(){i=r("p"),i.textContent=k,m=o(),p(c.$$.fragment)},l(s){i=a(s,"P",{"data-svelte-h":!0}),l(i)!=="svelte-11lpom8"&&(i.textContent=k),m=n(s),u(c.$$.fragment,s)},m(s,A){$(s,i,A),$(s,m,A),g(c,s,A),q=!0},p:z,i(s){q||(h(c.$$.fragment,s),q=!0)},o(s){f(c.$$.fragment,s),q=!1},d(s){s&&(d(i),d(m)),_(c,s)}}}function Pm(C){let i,k="Example:",m,c,q;return c=new ve({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMFRyYWluaW5nQXJndW1lbnRzJTBBJTBBYXJncyUyMCUzRCUyMFRyYWluaW5nQXJndW1lbnRzKCUyMndvcmtpbmdfZGlyJTIyKSUwQWFyZ3MlMjAlM0QlMjBhcmdzLnNldF9vcHRpbWl6ZXIobmFtZSUzRCUyMmFkYW13X3RvcmNoJTIyJTJDJTIwYmV0YTElM0QwLjgpJTBBYXJncy5vcHRpbQ==",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> TrainingArguments | |
| <span class="hljs-meta">>>> </span>args = TrainingArguments(<span class="hljs-string">"working_dir"</span>) | |
| <span class="hljs-meta">>>> </span>args = args.set_optimizer(name=<span class="hljs-string">"adamw_torch"</span>, beta1=<span class="hljs-number">0.8</span>) | |
| <span class="hljs-meta">>>> </span>args.optim | |
| <span class="hljs-string">'adamw_torch'</span>`,wrap:!1}}),{c(){i=r("p"),i.textContent=k,m=o(),p(c.$$.fragment)},l(s){i=a(s,"P",{"data-svelte-h":!0}),l(i)!=="svelte-11lpom8"&&(i.textContent=k),m=n(s),u(c.$$.fragment,s)},m(s,A){$(s,i,A),$(s,m,A),g(c,s,A),q=!0},p:z,i(s){q||(h(c.$$.fragment,s),q=!0)},o(s){f(c.$$.fragment,s),q=!1},d(s){s&&(d(i),d(m)),_(c,s)}}}function Mm(C){let i,k=`Calling this method will set <code>self.push_to_hub</code> to <code>True</code>, which means the <code>output_dir</code> will begin a git | |
| directory synced with the repo (determined by <code>model_id</code>) and the content will be pushed each time a save is | |
| triggered (depending on your <code>self.save_strategy</code>). Calling <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer.save_model">save_model()</a> will also trigger a push.`;return{c(){i=r("p"),i.innerHTML=k},l(m){i=a(m,"P",{"data-svelte-h":!0}),l(i)!=="svelte-65kum7"&&(i.innerHTML=k)},m(m,c){$(m,i,c)},p:z,d(m){m&&d(i)}}}function Fm(C){let i,k="Example:",m,c,q;return c=new ve({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMFRyYWluaW5nQXJndW1lbnRzJTBBJTBBYXJncyUyMCUzRCUyMFRyYWluaW5nQXJndW1lbnRzKCUyMndvcmtpbmdfZGlyJTIyKSUwQWFyZ3MlMjAlM0QlMjBhcmdzLnNldF9wdXNoX3RvX2h1YiglMjJtZSUyRmF3ZXNvbWUtbW9kZWwlMjIpJTBBYXJncy5odWJfbW9kZWxfaWQ=",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> TrainingArguments | |
| <span class="hljs-meta">>>> </span>args = TrainingArguments(<span class="hljs-string">"working_dir"</span>) | |
| <span class="hljs-meta">>>> </span>args = args.set_push_to_hub(<span class="hljs-string">"me/awesome-model"</span>) | |
| <span class="hljs-meta">>>> </span>args.hub_model_id | |
| <span class="hljs-string">'me/awesome-model'</span>`,wrap:!1}}),{c(){i=r("p"),i.textContent=k,m=o(),p(c.$$.fragment)},l(s){i=a(s,"P",{"data-svelte-h":!0}),l(i)!=="svelte-11lpom8"&&(i.textContent=k),m=n(s),u(c.$$.fragment,s)},m(s,A){$(s,i,A),$(s,m,A),g(c,s,A),q=!0},p:z,i(s){q||(h(c.$$.fragment,s),q=!0)},o(s){f(c.$$.fragment,s),q=!1},d(s){s&&(d(i),d(m)),_(c,s)}}}function zm(C){let i,k="Example:",m,c,q;return c=new ve({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMFRyYWluaW5nQXJndW1lbnRzJTBBJTBBYXJncyUyMCUzRCUyMFRyYWluaW5nQXJndW1lbnRzKCUyMndvcmtpbmdfZGlyJTIyKSUwQWFyZ3MlMjAlM0QlMjBhcmdzLnNldF9zYXZlKHN0cmF0ZWd5JTNEJTIyc3RlcHMlMjIlMkMlMjBzdGVwcyUzRDEwMCklMEFhcmdzLnNhdmVfc3RlcHM=",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> TrainingArguments | |
| <span class="hljs-meta">>>> </span>args = TrainingArguments(<span class="hljs-string">"working_dir"</span>) | |
| <span class="hljs-meta">>>> </span>args = args.set_save(strategy=<span class="hljs-string">"steps"</span>, steps=<span class="hljs-number">100</span>) | |
| <span class="hljs-meta">>>> </span>args.save_steps | |
| <span class="hljs-number">100</span>`,wrap:!1}}),{c(){i=r("p"),i.textContent=k,m=o(),p(c.$$.fragment)},l(s){i=a(s,"P",{"data-svelte-h":!0}),l(i)!=="svelte-11lpom8"&&(i.textContent=k),m=n(s),u(c.$$.fragment,s)},m(s,A){$(s,i,A),$(s,m,A),g(c,s,A),q=!0},p:z,i(s){q||(h(c.$$.fragment,s),q=!0)},o(s){f(c.$$.fragment,s),q=!1},d(s){s&&(d(i),d(m)),_(c,s)}}}function Dm(C){let i,k="Calling this method will automatically set <code>self.do_predict</code> to <code>True</code>.";return{c(){i=r("p"),i.innerHTML=k},l(m){i=a(m,"P",{"data-svelte-h":!0}),l(i)!=="svelte-ci4epi"&&(i.innerHTML=k)},m(m,c){$(m,i,c)},p:z,d(m){m&&d(i)}}}function Im(C){let i,k="Example:",m,c,q;return c=new ve({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMFRyYWluaW5nQXJndW1lbnRzJTBBJTBBYXJncyUyMCUzRCUyMFRyYWluaW5nQXJndW1lbnRzKCUyMndvcmtpbmdfZGlyJTIyKSUwQWFyZ3MlMjAlM0QlMjBhcmdzLnNldF90ZXN0aW5nKGJhdGNoX3NpemUlM0QzMiklMEFhcmdzLnBlcl9kZXZpY2VfZXZhbF9iYXRjaF9zaXpl",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> TrainingArguments | |
| <span class="hljs-meta">>>> </span>args = TrainingArguments(<span class="hljs-string">"working_dir"</span>) | |
| <span class="hljs-meta">>>> </span>args = args.set_testing(batch_size=<span class="hljs-number">32</span>) | |
| <span class="hljs-meta">>>> </span>args.per_device_eval_batch_size | |
| <span class="hljs-number">32</span>`,wrap:!1}}),{c(){i=r("p"),i.textContent=k,m=o(),p(c.$$.fragment)},l(s){i=a(s,"P",{"data-svelte-h":!0}),l(i)!=="svelte-11lpom8"&&(i.textContent=k),m=n(s),u(c.$$.fragment,s)},m(s,A){$(s,i,A),$(s,m,A),g(c,s,A),q=!0},p:z,i(s){q||(h(c.$$.fragment,s),q=!0)},o(s){f(c.$$.fragment,s),q=!1},d(s){s&&(d(i),d(m)),_(c,s)}}}function Um(C){let i,k="Calling this method will automatically set <code>self.do_train</code> to <code>True</code>.";return{c(){i=r("p"),i.innerHTML=k},l(m){i=a(m,"P",{"data-svelte-h":!0}),l(i)!=="svelte-102o7rb"&&(i.innerHTML=k)},m(m,c){$(m,i,c)},p:z,d(m){m&&d(i)}}}function Wm(C){let i,k="Example:",m,c,q;return c=new ve({props:{code:"ZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMFRyYWluaW5nQXJndW1lbnRzJTBBJTBBYXJncyUyMCUzRCUyMFRyYWluaW5nQXJndW1lbnRzKCUyMndvcmtpbmdfZGlyJTIyKSUwQWFyZ3MlMjAlM0QlMjBhcmdzLnNldF90cmFpbmluZyhsZWFybmluZ19yYXRlJTNEMWUtNCUyQyUyMGJhdGNoX3NpemUlM0QzMiklMEFhcmdzLmxlYXJuaW5nX3JhdGU=",highlighted:`<span class="hljs-meta">>>> </span><span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> TrainingArguments | |
| <span class="hljs-meta">>>> </span>args = TrainingArguments(<span class="hljs-string">"working_dir"</span>) | |
| <span class="hljs-meta">>>> </span>args = args.set_training(learning_rate=<span class="hljs-number">1e-4</span>, batch_size=<span class="hljs-number">32</span>) | |
| <span class="hljs-meta">>>> </span>args.learning_rate | |
| <span class="hljs-number">1e-4</span>`,wrap:!1}}),{c(){i=r("p"),i.textContent=k,m=o(),p(c.$$.fragment)},l(s){i=a(s,"P",{"data-svelte-h":!0}),l(i)!=="svelte-11lpom8"&&(i.textContent=k),m=n(s),u(c.$$.fragment,s)},m(s,A){$(s,i,A),$(s,m,A),g(c,s,A),q=!0},p:z,i(s){q||(h(c.$$.fragment,s),q=!0)},o(s){f(c.$$.fragment,s),q=!1},d(s){s&&(d(i),d(m)),_(c,s)}}}function Lm(C){let i,k,m,c,q,s,A,md='<a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a> 클래스는 PyTorch에서 완전한 기능(feature-complete)의 훈련을 위한 API를 제공하며, 다중 GPU/TPU에서의 분산 훈련, <a href="https://nvidia.github.io/apex/" rel="nofollow">NVIDIA GPU</a>, <a href="https://rocm.docs.amd.com/en/latest/rocm.html" rel="nofollow">AMD GPU</a>를 위한 혼합 정밀도, 그리고 PyTorch의 <a href="https://pytorch.org/docs/stable/amp.html" rel="nofollow"><code>torch.amp</code></a>를 지원합니다. <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>는 모델의 훈련 방식을 커스터마이즈할 수 있는 다양한 옵션을 제공하는 <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.TrainingArguments">TrainingArguments</a> 클래스와 함께 사용됩니다. 이 두 클래스는 함께 완전한 훈련 API를 제공합니다.',ga,Tt,pd='<a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Seq2SeqTrainer">Seq2SeqTrainer</a>와 <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Seq2SeqTrainingArguments">Seq2SeqTrainingArguments</a>는 <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>와 <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.TrainingArguments">TrainingArguments</a> 클래스를 상속하며, 요약이나 번역과 같은 시퀀스-투-시퀀스 작업을 위한 모델 훈련에 적합하게 조정되어 있습니다.',ha,$e,fa,wt,_a,b,xt,Ra,Bo,ud="Trainer is a simple but feature-complete training and eval loop for PyTorch, optimized for 🤗 Transformers.",Ja,Go,gd="Important attributes:",Ea,Vo,hd=`<li><strong>model</strong> — Always points to the core model. If using a transformers model, it will be a <code>PreTrainedModel</code> | |
| subclass.</li> <li><strong>model_wrapped</strong> — Always points to the most external model in case one or more other modules wrap the | |
| original model. This is the model that should be used for the forward pass. For example, under <code>DeepSpeed</code>, | |
| the inner model is wrapped in <code>DeepSpeed</code> and then again in <code>torch.nn.DistributedDataParallel</code>. If the inner | |
| model hasn’t been wrapped, then <code>self.model_wrapped</code> is the same as <code>self.model</code>.</li> <li><strong>is_model_parallel</strong> — Whether or not a model has been switched to a model parallel mode (different from | |
| data parallelism, this means some of the model layers are split on different GPUs).</li> <li><strong>place_model_on_device</strong> — Whether or not to automatically place the model on the device - it will be set | |
| to <code>False</code> if model parallel or deepspeed is used, or if the default | |
| <code>TrainingArguments.place_model_on_device</code> is overridden to return <code>False</code> .</li> <li><strong>is_in_train</strong> — Whether or not a model is currently running <code>train</code> (e.g. when <code>evaluate</code> is called while | |
| in <code>train</code>)</li>`,Ba,Ae,qt,Ga,Xo,fd="Add a callback to the current list of <code>TrainerCallback</code>.",Va,Se,kt,Xa,Zo,_d=`A helper wrapper that creates an appropriate context manager for <code>autocast</code> while feeding it the desired | |
| arguments, depending on the situation.`,Za,X,$t,Ya,Yo,bd="How the loss is computed by Trainer. By default, all models return the loss in the first element.",Qa,Qo,vd="Subclass and override for custom behavior.",Ka,Ce,At,es,Ko,yd="A helper wrapper to group together context managers.",ts,Pe,St,os,en,Td="Creates a draft of a model card using the information available to the <code>Trainer</code>.",ns,Z,Ct,rs,tn,wd="Setup the optimizer.",as,on,xd=`We provide a reasonable default that works well. If you want to use something else, you can pass a tuple in the | |
| Trainer’s init through <code>optimizers</code>, or subclass and override this method in a subclass.`,ss,Y,Pt,is,nn,qd="Setup the optimizer and the learning rate scheduler.",ls,rn,kd=`We provide a reasonable default that works well. If you want to use something else, you can pass a tuple in the | |
| Trainer’s init through <code>optimizers</code>, or subclass and override this method (or <code>create_optimizer</code> and/or | |
| <code>create_scheduler</code>) in a subclass.`,ds,Me,Mt,cs,an,$d=`Setup the scheduler. The optimizer of the trainer must have been set up either before this method is called or | |
| passed as an argument.`,ms,L,Ft,ps,sn,Ad="Run evaluation and returns metrics.",us,ln,Sd=`The calling script will be responsible for providing a method to compute metrics, as they are task-dependent | |
| (pass it to the init <code>compute_metrics</code> argument).`,gs,dn,Cd="You can also subclass and override this method to inject custom behavior.",hs,Q,zt,fs,cn,Pd="Prediction/evaluation loop, shared by <code>Trainer.evaluate()</code> and <code>Trainer.predict()</code>.",_s,mn,Md="Works both with or without labels.",bs,Fe,Dt,vs,pn,Fd=`For models that inherit from <code>PreTrainedModel</code>, uses that method to compute the number of floating point | |
| operations for every backward + forward pass. If using another model, either implement such a method in the | |
| model or subclass and override this method.`,ys,K,It,Ts,un,zd="Get all parameter names that weight decay will be applied to",ws,gn,Dd=`Note that some models implement their own layernorm instead of calling nn.LayerNorm, weight decay could still | |
| apply to those modules since this function only filter out instance of nn.LayerNorm`,xs,ee,Ut,qs,hn,Id="Returns the evaluation <code>~torch.utils.data.DataLoader</code>.",ks,fn,Ud="Subclass and override this method if you want to inject some custom behavior.",$s,ze,Wt,As,_n,Wd="Returns the learning rate of each parameter from self.optimizer.",Ss,De,Lt,Cs,bn,Ld="Get the number of trainable parameters.",Ps,Ie,Nt,Ms,vn,Nd="Returns the optimizer class and optimizer parameters based on the training arguments.",Fs,Ue,jt,zs,yn,jd="Returns optimizer group for a parameter if given, else returns all optimizer groups for params.",Ds,te,Ot,Is,Tn,Od="Returns the test <code>~torch.utils.data.DataLoader</code>.",Us,wn,Hd="Subclass and override this method if you want to inject some custom behavior.",Ws,N,Ht,Ls,xn,Rd="Returns the training <code>~torch.utils.data.DataLoader</code>.",Ns,qn,Jd=`Will use no sampler if <code>train_dataset</code> does not implement <code>__len__</code>, a random sampler (adapted to distributed | |
| training if necessary) otherwise.`,js,kn,Ed="Subclass and override this method if you want to inject some custom behavior.",Os,oe,Rt,Hs,$n,Bd=`Launch an hyperparameter search using <code>optuna</code> or <code>Ray Tune</code> or <code>SigOpt</code>. The optimized quantity is determined | |
| by <code>compute_objective</code>, which defaults to a function returning the evaluation loss when no metric is provided, | |
| the sum of all metrics otherwise.`,Rs,We,Js,Le,Jt,Es,An,Gd="Initializes a git repo in <code>self.args.hub_model_id</code>.",Bs,Ne,Et,Gs,Sn,Vd=`Whether or not this process is the local (e.g., on one machine if training in a distributed fashion on several | |
| machines) main process.`,Vs,je,Bt,Xs,Cn,Xd=`Whether or not this process is the global main process (when training in a distributed fashion on several | |
| machines, this is only going to be <code>True</code> for one process).`,Zs,ne,Gt,Ys,Pn,Zd="Log <code>logs</code> on the various objects watching training.",Qs,Mn,Yd="Subclass and override this method to inject custom behavior.",Ks,P,Vt,ei,Fn,Qd="Log metrics in a specially formatted way",ti,zn,Kd="Under distributed environment this is done only for a process with rank 0.",oi,Dn,ec="Notes on memory reports:",ni,In,tc="In order to get memory usage report you need to install <code>psutil</code>. You can do that with <code>pip install psutil</code>.",ri,Oe,ai,Un,oc="<strong>Understanding the reports:</strong>",si,Wn,nc=`<li>the first segment, e.g., <code>train__</code>, tells you which stage the metrics are for. Reports starting with <code>init_</code> | |
| will be added to the first stage that gets run. So that if only evaluation is run, the memory usage for the | |
| <code>__init__</code> will be reported along with the <code>eval_</code> metrics.</li> <li>the third segment, is either <code>cpu</code> or <code>gpu</code>, tells you whether it’s the general RAM or the gpu0 memory | |
| metric.</li> <li><code>*_alloc_delta</code> - is the difference in the used/allocated memory counter between the end and the start of the | |
| stage - it can be negative if a function released more memory than it allocated.</li> <li><code>*_peaked_delta</code> - is any extra memory that was consumed and then freed - relative to the current allocated | |
| memory counter - it is never negative. When you look at the metrics of any stage you add up <code>alloc_delta</code> + | |
| <code>peaked_delta</code> and you know how much memory was needed to complete that stage.</li>`,ii,Ln,rc=`The reporting happens only for process of rank 0 and gpu 0 (if there is a gpu). Typically this is enough since the | |
| main process does the bulk of work, but it could be not quite so if model parallel is used and then other GPUs may | |
| use a different amount of gpu memory. This is also not the same under DataParallel where gpu0 may require much more | |
| memory than the rest since it stores the gradient and optimizer states for all participating GPUS. Perhaps in the | |
| future these reports will evolve to measure those too.`,li,Nn,ac=`The CPU RAM metric measures RSS (Resident Set Size) includes both the memory which is unique to the process and the | |
| memory shared with other processes. It is important to note that it does not include swapped out memory, so the | |
| reports could be imprecise.`,di,jn,sc=`The CPU peak memory is measured using a sampling thread. Due to python’s GIL it may miss some of the peak memory if | |
| that thread didn’t get a chance to run when the highest memory was used. Therefore this report can be less than | |
| reality. Using <code>tracemalloc</code> would have reported the exact peak memory, but it doesn’t report memory allocations | |
| outside of python. So if some C++ CUDA extension allocated its own memory it won’t be reported. And therefore it | |
| was dropped in favor of the memory sampling approach, which reads the current process memory usage.`,ci,On,ic=`The GPU allocated and peak memory reporting is done with <code>torch.cuda.memory_allocated()</code> and | |
| <code>torch.cuda.max_memory_allocated()</code>. This metric reports only “deltas” for pytorch-specific allocations, as | |
| <code>torch.cuda</code> memory management system doesn’t track any memory allocated outside of pytorch. For example, the very | |
| first cuda call typically loads CUDA kernels, which may take from 0.5 to 2GB of GPU memory.`,mi,Hn,lc=`Note that this tracker doesn’t account for memory allocations outside of <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>’s <code>__init__</code>, <code>train</code>, | |
| <code>evaluate</code> and <code>predict</code> calls.`,pi,Rn,dc=`Because <code>evaluation</code> calls may happen during <code>train</code>, we can’t handle nested invocations because | |
| <code>torch.cuda.max_memory_allocated</code> is a single counter, so if it gets reset by a nested eval call, <code>train</code>’s tracker | |
| will report incorrect info. If this <a href="https://github.com/pytorch/pytorch/issues/16266" rel="nofollow">pytorch issue</a> gets resolved | |
| it will be possible to change this class to be re-entrant. Until then we will only track the outer level of | |
| <code>train</code>, <code>evaluate</code> and <code>predict</code> methods. Which means that if <code>eval</code> is called during <code>train</code>, it’s the latter | |
| that will account for its memory usage and that of the former.`,ui,Jn,cc=`This also means that if any other tool that is used along the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a> calls | |
| <code>torch.cuda.reset_peak_memory_stats</code>, the gpu peak memory stats could be invalid. And the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a> will disrupt | |
| the normal behavior of any such tools that rely on calling <code>torch.cuda.reset_peak_memory_stats</code> themselves.`,gi,En,mc="For best performance you may want to consider turning the memory profiling off for production runs.",hi,He,Xt,fi,Bn,pc="Reformat Trainer metrics values to a human-readable format",_i,Re,Zt,bi,Gn,uc=`Helper to get number of samples in a <code>~torch.utils.data.DataLoader</code> by accessing its dataset. When | |
| dataloader.dataset does not exist or has no length, estimates as best it can`,vi,Je,Yt,yi,Vn,gc="Helper to get number of tokens in a <code>~torch.utils.data.DataLoader</code> by enumerating dataloader.",Ti,re,Qt,wi,Xn,hc="Remove a callback from the current list of <code>TrainerCallback</code> and returns it.",xi,Zn,fc="If the callback is not found, returns <code>None</code> (and no error is raised).",qi,D,Kt,ki,Yn,_c="Run prediction and returns predictions and potential metrics.",$i,Qn,bc=`Depending on the dataset and your use case, your test dataset may contain labels. In that case, this method | |
| will also return metrics, like in <code>evaluate()</code>.`,Ai,Ee,Si,Kn,vc="Returns: <em>NamedTuple</em> A namedtuple with the following keys:",Ci,er,yc=`<li>predictions (<code>np.ndarray</code>): The predictions on <code>test_dataset</code>.</li> <li>label_ids (<code>np.ndarray</code>, <em>optional</em>): The labels (if the dataset contained some).</li> <li>metrics (<code>Dict[str, float]</code>, <em>optional</em>): The potential dictionary of metrics (if the dataset contained | |
| labels).</li>`,Pi,ae,eo,Mi,tr,Tc="Prediction/evaluation loop, shared by <code>Trainer.evaluate()</code> and <code>Trainer.predict()</code>.",Fi,or,wc="Works both with or without labels.",zi,se,to,Di,nr,xc="Perform an evaluation step on <code>model</code> using <code>inputs</code>.",Ii,rr,qc="Subclass and override to inject custom behavior.",Ui,Be,oo,Wi,ar,kc="Sets values in the deepspeed plugin based on the Trainer args",Li,Ge,no,Ni,sr,$c="Upload <code>self.model</code> and <code>self.processing_class</code> to the 🤗 model hub on the repo <code>self.args.hub_model_id</code>.",ji,Ve,ro,Oi,ir,Ac="Remove a callback from the current list of <code>TrainerCallback</code>.",Hi,j,ao,Ri,lr,Sc="Save metrics into a json file for that split, e.g. <code>train_results.json</code>.",Ji,dr,Cc="Under distributed environment this is done only for a process with rank 0.",Ei,cr,Pc=`To understand the metrics please read the docstring of <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer.log_metrics">log_metrics()</a>. The only difference is that raw | |
| unformatted numbers are saved in the current method.`,Bi,ie,so,Gi,mr,Mc="Will save the model, so you can reload it using <code>from_pretrained()</code>.",Vi,pr,Fc="Will only save from the main process.",Xi,le,io,Zi,ur,zc="Saves the Trainer state, since Trainer.save_model saves only the tokenizer with the model",Yi,gr,Dc="Under distributed environment this is done only for a process with rank 0.",Qi,Xe,lo,Ki,hr,Ic="Main training entry point.",el,de,co,tl,fr,Uc="Perform a training step on a batch of inputs.",ol,_r,Wc="Subclass and override to inject custom behavior.",ba,mo,va,G,po,nl,O,uo,rl,br,Lc="Run evaluation and returns metrics.",al,vr,Nc=`The calling script will be responsible for providing a method to compute metrics, as they are task-dependent | |
| (pass it to the init <code>compute_metrics</code> argument).`,sl,yr,jc="You can also subclass and override this method to inject custom behavior.",il,I,go,ll,Tr,Oc="Run prediction and returns predictions and potential metrics.",dl,wr,Hc=`Depending on the dataset and your use case, your test dataset may contain labels. In that case, this method | |
| will also return metrics, like in <code>evaluate()</code>.`,cl,Ze,ml,xr,Rc="Returns: <em>NamedTuple</em> A namedtuple with the following keys:",pl,qr,Jc=`<li>predictions (<code>np.ndarray</code>): The predictions on <code>test_dataset</code>.</li> <li>label_ids (<code>np.ndarray</code>, <em>optional</em>): The labels (if the dataset contained some).</li> <li>metrics (<code>Dict[str, float]</code>, <em>optional</em>): The potential dictionary of metrics (if the dataset contained | |
| labels).</li>`,ya,ho,Ta,S,fo,ul,kr,Ec=`TrainingArguments is the subset of the arguments we use in our example scripts <strong>which relate to the training loop | |
| itself</strong>.`,gl,$r,Bc=`Using <code>HfArgumentParser</code> we can turn this class into | |
| <a href="https://docs.python.org/3/library/argparse#module-argparse" rel="nofollow">argparse</a> arguments that can be specified on the | |
| command line.`,hl,U,_o,fl,Ar,Gc=`Returns the log level to be used depending on whether this process is the main process of node 0, main process | |
| of node non-0, or a non-main process.`,_l,Sr,Vc=`For the main process the log level defaults to the logging level set (<code>logging.WARNING</code> if you didn’t do | |
| anything) unless overridden by <code>log_level</code> argument.`,bl,Cr,Xc=`For the replica processes the log level defaults to <code>logging.WARNING</code> unless overridden by <code>log_level_replica</code> | |
| argument.`,vl,Pr,Zc="The choice between the main and replica process settings is made according to the return value of <code>should_log</code>.",yl,Ye,bo,Tl,Mr,Yc="Get number of steps used for a linear warmup.",wl,ce,vo,xl,Fr,Qc=`A context manager for torch distributed environment where on needs to do something on the main process, while | |
| blocking replicas, and when it’s finished releasing the replicas.`,ql,zr,Kc=`One such use is for <code>datasets</code>’s <code>map</code> feature which to be efficient should be run once on the main process, | |
| which upon completion saves a cached version of results and which then automatically gets loaded by the | |
| replicas.`,kl,me,yo,$l,Dr,em="A method that regroups all arguments linked to the dataloaders creation.",Al,Qe,Sl,pe,To,Cl,Ir,tm="A method that regroups all arguments linked to evaluation.",Pl,Ke,Ml,ue,wo,Fl,Ur,om="A method that regroups all arguments linked to logging.",zl,et,Dl,ge,xo,Il,Wr,nm="A method that regroups all arguments linked to the learning rate scheduler and its hyperparameters.",Ul,tt,Wl,he,qo,Ll,Lr,rm="A method that regroups all arguments linked to the optimizer and its hyperparameters.",Nl,ot,jl,H,ko,Ol,Nr,am="A method that regroups all arguments linked to synchronizing checkpoints with the Hub.",Hl,nt,Rl,rt,Jl,fe,$o,El,jr,sm="A method that regroups all arguments linked to checkpoint saving.",Bl,at,Gl,R,Ao,Vl,Or,im="A method that regroups all basic arguments linked to testing on a held-out dataset.",Xl,st,Zl,it,Yl,J,So,Ql,Hr,lm="A method that regroups all basic arguments linked to the training.",Kl,lt,ed,dt,td,ct,Co,od,Rr,dm=`Serializes this instance while replace <code>Enum</code> by their values (for JSON serialization support). It obfuscates | |
| the token values by removing their value.`,nd,mt,Po,rd,Jr,cm="Serializes this instance to a JSON string.",ad,pt,Mo,sd,Er,mm="Sanitized serialization to use with TensorBoard’s hparams",wa,Fo,xa,W,zo,id,Br,pm=`TrainingArguments is the subset of the arguments we use in our example scripts <strong>which relate to the training loop | |
| itself</strong>.`,ld,Gr,um=`Using <code>HfArgumentParser</code> we can turn this class into | |
| <a href="https://docs.python.org/3/library/argparse#module-argparse" rel="nofollow">argparse</a> arguments that can be specified on the | |
| command line.`,dd,ut,Do,cd,Vr,gm=`Serializes this instance while replace <code>Enum</code> by their values and <code>GenerationConfig</code> by dictionaries (for JSON | |
| serialization support). It obfuscates the token values by removing their value.`,qa,Io,ka,pa,$a;return q=new ua({props:{title:"Trainer",local:"trainer",headingTag:"h1"}}),$e=new Eo({props:{warning:!0,$$slots:{default:[Tm]},$$scope:{ctx:C}}}),wt=new ua({props:{title:"Trainer",local:"transformers.Trainer ][ transformers.Trainer",headingTag:"h2"}}),xt=new x({props:{name:"class transformers.Trainer",anchor:"transformers.Trainer",parameters:[{name:"model",val:": Union = None"},{name:"args",val:": TrainingArguments = None"},{name:"data_collator",val:": Optional = None"},{name:"train_dataset",val:": Union = None"},{name:"eval_dataset",val:": Union = None"},{name:"processing_class",val:": Union = None"},{name:"model_init",val:": Optional = None"},{name:"compute_metrics",val:": Optional = None"},{name:"callbacks",val:": Optional = None"},{name:"optimizers",val:": Tuple = (None, None)"},{name:"preprocess_logits_for_metrics",val:": Optional = None"}],parametersDescription:[{anchor:"transformers.Trainer.model",description:`<strong>model</strong> (<code>PreTrainedModel</code> or <code>torch.nn.Module</code>, <em>optional</em>) — | |
| The model to train, evaluate or use for predictions. If not provided, a <code>model_init</code> must be passed.</p> | |
| <div class="course-tip bg-gradient-to-br dark:bg-gradient-to-r before:border-green-500 dark:before:border-green-800 from-green-50 dark:from-gray-900 to-white dark:to-gray-950 border border-green-50 text-green-700 dark:text-gray-400"> | |
| <p><a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a> is optimized to work with the <code>PreTrainedModel</code> provided by the library. You can still use | |
| your own models defined as <code>torch.nn.Module</code> as long as they work the same way as the 🤗 Transformers | |
| models.</p> | |
| </div>`,name:"model"},{anchor:"transformers.Trainer.args",description:`<strong>args</strong> (<a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.TrainingArguments">TrainingArguments</a>, <em>optional</em>) — | |
| The arguments to tweak for training. Will default to a basic instance of <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.TrainingArguments">TrainingArguments</a> with the | |
| <code>output_dir</code> set to a directory named <em>tmp_trainer</em> in the current directory if not provided.`,name:"args"},{anchor:"transformers.Trainer.data_collator",description:`<strong>data_collator</strong> (<code>DataCollator</code>, <em>optional</em>) — | |
| The function to use to form a batch from a list of elements of <code>train_dataset</code> or <code>eval_dataset</code>. Will | |
| default to <code>default_data_collator()</code> if no <code>processing_class</code> is provided, an instance of | |
| <code>DataCollatorWithPadding</code> otherwise if the processing_class is a feature extractor or tokenizer.`,name:"data_collator"},{anchor:"transformers.Trainer.train_dataset",description:`<strong>train_dataset</strong> (Union[<code>torch.utils.data.Dataset</code>, <code>torch.utils.data.IterableDataset</code>, <code>datasets.Dataset</code>], <em>optional</em>) — | |
| The dataset to use for training. If it is a <code>Dataset</code>, columns not accepted by the | |
| <code>model.forward()</code> method are automatically removed.</p> | |
| <p>Note that if it’s a <code>torch.utils.data.IterableDataset</code> with some randomization and you are training in a | |
| distributed fashion, your iterable dataset should either use a internal attribute <code>generator</code> that is a | |
| <code>torch.Generator</code> for the randomization that must be identical on all processes (and the Trainer will | |
| manually set the seed of this <code>generator</code> at each epoch) or have a <code>set_epoch()</code> method that internally | |
| sets the seed of the RNGs used.`,name:"train_dataset"},{anchor:"transformers.Trainer.eval_dataset",description:`<strong>eval_dataset</strong> (Union[<code>torch.utils.data.Dataset</code>, Dict[str, <code>torch.utils.data.Dataset</code>, <code>datasets.Dataset</code>]), <em>optional</em>) — | |
| The dataset to use for evaluation. If it is a <code>Dataset</code>, columns not accepted by the | |
| <code>model.forward()</code> method are automatically removed. If it is a dictionary, it will evaluate on each | |
| dataset prepending the dictionary key to the metric name.`,name:"eval_dataset"},{anchor:"transformers.Trainer.processing_class",description:`<strong>processing_class</strong> (<code>PreTrainedTokenizerBase</code> or <code>BaseImageProcessor</code> or <code>FeatureExtractionMixin</code> or <code>ProcessorMixin</code>, <em>optional</em>) — | |
| Processing class used to process the data. If provided, will be used to automatically process the inputs | |
| for the model, and it will be saved along the model to make it easier to rerun an interrupted training or | |
| reuse the fine-tuned model. | |
| This supercedes the <code>tokenizer</code> argument, which is now deprecated.`,name:"processing_class"},{anchor:"transformers.Trainer.model_init",description:`<strong>model_init</strong> (<code>Callable[[], PreTrainedModel]</code>, <em>optional</em>) — | |
| A function that instantiates the model to be used. If provided, each call to <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer.train">train()</a> will start | |
| from a new instance of the model as given by this function.</p> | |
| <p>The function may have zero argument, or a single one containing the optuna/Ray Tune/SigOpt trial object, to | |
| be able to choose different architectures according to hyper parameters (such as layer count, sizes of | |
| inner layers, dropout probabilities etc).`,name:"model_init"},{anchor:"transformers.Trainer.compute_metrics",description:`<strong>compute_metrics</strong> (<code>Callable[[EvalPrediction], Dict]</code>, <em>optional</em>) — | |
| The function that will be used to compute metrics at evaluation. Must take a <code>EvalPrediction</code> and return | |
| a dictionary string to metric values. <em>Note</em> When passing TrainingArgs with <code>batch_eval_metrics</code> set to | |
| <code>True</code>, your compute_metrics function must take a boolean <code>compute_result</code> argument. This will be triggered | |
| after the last eval batch to signal that the function needs to calculate and return the global summary | |
| statistics rather than accumulating the batch-level statistics.`,name:"compute_metrics"},{anchor:"transformers.Trainer.callbacks",description:`<strong>callbacks</strong> (List of <code>TrainerCallback</code>, <em>optional</em>) — | |
| A list of callbacks to customize the training loop. Will add those to the list of default callbacks | |
| detailed in <a href="callback">here</a>.</p> | |
| <p>If you want to remove one of the default callbacks used, use the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer.remove_callback">Trainer.remove_callback()</a> method.`,name:"callbacks"},{anchor:"transformers.Trainer.optimizers",description:`<strong>optimizers</strong> (<code>Tuple[torch.optim.Optimizer, torch.optim.lr_scheduler.LambdaLR]</code>, <em>optional</em>, defaults to <code>(None, None)</code>) — | |
| A tuple containing the optimizer and the scheduler to use. Will default to an instance of <code>AdamW</code> on your | |
| model and a scheduler given by <code>get_linear_schedule_with_warmup()</code> controlled by <code>args</code>.`,name:"optimizers"},{anchor:"transformers.Trainer.preprocess_logits_for_metrics",description:`<strong>preprocess_logits_for_metrics</strong> (<code>Callable[[torch.Tensor, torch.Tensor], torch.Tensor]</code>, <em>optional</em>) — | |
| A function that preprocess the logits right before caching them at each evaluation step. Must take two | |
| tensors, the logits and the labels, and return the logits once processed as desired. The modifications made | |
| by this function will be reflected in the predictions received by <code>compute_metrics</code>.</p> | |
| <p>Note that the labels (second parameter) will be <code>None</code> if the dataset does not have them.`,name:"preprocess_logits_for_metrics"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L295"}}),qt=new x({props:{name:"add_callback",anchor:"transformers.Trainer.add_callback",parameters:[{name:"callback",val:""}],parametersDescription:[{anchor:"transformers.Trainer.add_callback.callback",description:"<strong>callback</strong> (<code>type</code> or [`~transformers.TrainerCallback]<code>) -- A </code>TrainerCallback<code>class or an instance of a</code>TrainerCallback`. In the\nfirst case, will instantiate a member of that class.",name:"callback"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L789"}}),kt=new x({props:{name:"autocast_smart_context_manager",anchor:"transformers.Trainer.autocast_smart_context_manager",parameters:[{name:"cache_enabled",val:": Optional = True"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L3483"}}),$t=new x({props:{name:"compute_loss",anchor:"transformers.Trainer.compute_loss",parameters:[{name:"model",val:""},{name:"inputs",val:""},{name:"return_outputs",val:" = False"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L3560"}}),At=new x({props:{name:"compute_loss_context_manager",anchor:"transformers.Trainer.compute_loss_context_manager",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L3477"}}),St=new x({props:{name:"create_model_card",anchor:"transformers.Trainer.create_model_card",parameters:[{name:"language",val:": Optional = None"},{name:"license",val:": Optional = None"},{name:"tags",val:": Union = None"},{name:"model_name",val:": Optional = None"},{name:"finetuned_from",val:": Optional = None"},{name:"tasks",val:": Union = None"},{name:"dataset_tags",val:": Union = None"},{name:"dataset",val:": Union = None"},{name:"dataset_args",val:": Union = None"}],parametersDescription:[{anchor:"transformers.Trainer.create_model_card.language",description:`<strong>language</strong> (<code>str</code>, <em>optional</em>) — | |
| The language of the model (if applicable)`,name:"language"},{anchor:"transformers.Trainer.create_model_card.license",description:`<strong>license</strong> (<code>str</code>, <em>optional</em>) — | |
| The license of the model. Will default to the license of the pretrained model used, if the original | |
| model given to the <code>Trainer</code> comes from a repo on the Hub.`,name:"license"},{anchor:"transformers.Trainer.create_model_card.tags",description:`<strong>tags</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) — | |
| Some tags to be included in the metadata of the model card.`,name:"tags"},{anchor:"transformers.Trainer.create_model_card.model_name",description:`<strong>model_name</strong> (<code>str</code>, <em>optional</em>) — | |
| The name of the model.`,name:"model_name"},{anchor:"transformers.Trainer.create_model_card.finetuned_from",description:`<strong>finetuned_from</strong> (<code>str</code>, <em>optional</em>) — | |
| The name of the model used to fine-tune this one (if applicable). Will default to the name of the repo | |
| of the original model given to the <code>Trainer</code> (if it comes from the Hub).`,name:"finetuned_from"},{anchor:"transformers.Trainer.create_model_card.tasks",description:`<strong>tasks</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) — | |
| One or several task identifiers, to be included in the metadata of the model card.`,name:"tasks"},{anchor:"transformers.Trainer.create_model_card.dataset_tags",description:`<strong>dataset_tags</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) — | |
| One or several dataset tags, to be included in the metadata of the model card.`,name:"dataset_tags"},{anchor:"transformers.Trainer.create_model_card.dataset",description:`<strong>dataset</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) — | |
| One or several dataset identifiers, to be included in the metadata of the model card.`,name:"dataset"},{anchor:"transformers.Trainer.create_model_card.dataset_args",description:`<strong>dataset_args</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>) — | |
| One or several dataset arguments, to be included in the metadata of the model card.`,name:"dataset_args"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L4380"}}),Ct=new x({props:{name:"create_optimizer",anchor:"transformers.Trainer.create_optimizer",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L1104"}}),Pt=new x({props:{name:"create_optimizer_and_scheduler",anchor:"transformers.Trainer.create_optimizer_and_scheduler",parameters:[{name:"num_training_steps",val:": int"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L1077"}}),Mt=new x({props:{name:"create_scheduler",anchor:"transformers.Trainer.create_scheduler",parameters:[{name:"num_training_steps",val:": int"},{name:"optimizer",val:": Optimizer = None"}],parametersDescription:[{anchor:"transformers.Trainer.create_scheduler.num_training_steps",description:"<strong>num_training_steps</strong> (int) — The number of training steps to do.",name:"num_training_steps"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L1584"}}),Ft=new x({props:{name:"evaluate",anchor:"transformers.Trainer.evaluate",parameters:[{name:"eval_dataset",val:": Union = None"},{name:"ignore_keys",val:": Optional = None"},{name:"metric_key_prefix",val:": str = 'eval'"}],parametersDescription:[{anchor:"transformers.Trainer.evaluate.eval_dataset",description:`<strong>eval_dataset</strong> (Union[<code>Dataset</code>, Dict[str, <code>Dataset</code>]), <em>optional</em>) — | |
| Pass a dataset if you wish to override <code>self.eval_dataset</code>. If it is a <code>Dataset</code>, columns | |
| not accepted by the <code>model.forward()</code> method are automatically removed. If it is a dictionary, it will | |
| evaluate on each dataset, prepending the dictionary key to the metric name. Datasets must implement the | |
| <code>__len__</code> method.</p> | |
| <div class="course-tip bg-gradient-to-br dark:bg-gradient-to-r before:border-green-500 dark:before:border-green-800 from-green-50 dark:from-gray-900 to-white dark:to-gray-950 border border-green-50 text-green-700 dark:text-gray-400"> | |
| <p>If you pass a dictionary with names of datasets as keys and datasets as values, evaluate will run | |
| separate evaluations on each dataset. This can be useful to monitor how training affects other | |
| datasets or simply to get a more fine-grained evaluation. | |
| When used with <code>load_best_model_at_end</code>, make sure <code>metric_for_best_model</code> references exactly one | |
| of the datasets. If you, for example, pass in <code>{"data1": data1, "data2": data2}</code> for two datasets | |
| <code>data1</code> and <code>data2</code>, you could specify <code>metric_for_best_model="eval_data1_loss"</code> for using the | |
| loss on <code>data1</code> and <code>metric_for_best_model="eval_data2_loss"</code> for the loss on <code>data2</code>.</p> | |
| </div>`,name:"eval_dataset"},{anchor:"transformers.Trainer.evaluate.ignore_keys",description:`<strong>ignore_keys</strong> (<code>List[str]</code>, <em>optional</em>) — | |
| A list of keys in the output of your model (if it is a dictionary) that should be ignored when | |
| gathering predictions.`,name:"ignore_keys"},{anchor:"transformers.Trainer.evaluate.metric_key_prefix",description:`<strong>metric_key_prefix</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"eval"</code>) — | |
| An optional prefix to be used as the metrics key prefix. For example the metrics “bleu” will be named | |
| “eval_bleu” if the prefix is “eval” (default)`,name:"metric_key_prefix"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L3838",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A dictionary containing the evaluation loss and the potential metrics computed from the predictions. The | |
| dictionary also contains the epoch number which comes from the training state.</p> | |
| `}}),zt=new x({props:{name:"evaluation_loop",anchor:"transformers.Trainer.evaluation_loop",parameters:[{name:"dataloader",val:": DataLoader"},{name:"description",val:": str"},{name:"prediction_loss_only",val:": Optional = None"},{name:"ignore_keys",val:": Optional = None"},{name:"metric_key_prefix",val:": str = 'eval'"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L4006"}}),Dt=new x({props:{name:"floating_point_ops",anchor:"transformers.Trainer.floating_point_ops",parameters:[{name:"inputs",val:": Dict"}],parametersDescription:[{anchor:"transformers.Trainer.floating_point_ops.inputs",description:`<strong>inputs</strong> (<code>Dict[str, Union[torch.Tensor, Any]]</code>) — | |
| The inputs and targets of the model.`,name:"inputs"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L4344",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The number of floating-point operations.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>int</code></p> | |
| `}}),It=new x({props:{name:"get_decay_parameter_names",anchor:"transformers.Trainer.get_decay_parameter_names",parameters:[{name:"model",val:""}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L1093"}}),Ut=new x({props:{name:"get_eval_dataloader",anchor:"transformers.Trainer.get_eval_dataloader",parameters:[{name:"eval_dataset",val:": Union = None"}],parametersDescription:[{anchor:"transformers.Trainer.get_eval_dataloader.eval_dataset",description:`<strong>eval_dataset</strong> (<code>str</code> or <code>torch.utils.data.Dataset</code>, <em>optional</em>) — | |
| If a <code>str</code>, will use <code>self.eval_dataset[eval_dataset]</code> as the evaluation dataset. If a <code>Dataset</code>, will override <code>self.eval_dataset</code> and must implement <code>__len__</code>. If it is a <code>Dataset</code>, columns not accepted by the <code>model.forward()</code> method are automatically removed.`,name:"eval_dataset"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L982"}}),Wt=new x({props:{name:"get_learning_rates",anchor:"transformers.Trainer.get_learning_rates",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L1174"}}),Lt=new x({props:{name:"get_num_trainable_parameters",anchor:"transformers.Trainer.get_num_trainable_parameters",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L1168"}}),Nt=new x({props:{name:"get_optimizer_cls_and_kwargs",anchor:"transformers.Trainer.get_optimizer_cls_and_kwargs",parameters:[{name:"args",val:": TrainingArguments"},{name:"model",val:": Optional = None"}],parametersDescription:[{anchor:"transformers.Trainer.get_optimizer_cls_and_kwargs.args",description:`<strong>args</strong> (<code>transformers.training_args.TrainingArguments</code>) — | |
| The training arguments for the training session.`,name:"args"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L1198"}}),jt=new x({props:{name:"get_optimizer_group",anchor:"transformers.Trainer.get_optimizer_group",parameters:[{name:"param",val:": Union = None"}],parametersDescription:[{anchor:"transformers.Trainer.get_optimizer_group.param",description:`<strong>param</strong> (<code>str</code> or <code>torch.nn.parameter.Parameter</code>, <em>optional</em>) — | |
| The parameter for which optimizer group needs to be returned.`,name:"param"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L1182"}}),Ot=new x({props:{name:"get_test_dataloader",anchor:"transformers.Trainer.get_test_dataloader",parameters:[{name:"test_dataset",val:": Dataset"}],parametersDescription:[{anchor:"transformers.Trainer.get_test_dataloader.test_dataset",description:`<strong>test_dataset</strong> (<code>torch.utils.data.Dataset</code>, <em>optional</em>) — | |
| The test dataset to use. If it is a <code>Dataset</code>, columns not accepted by the | |
| <code>model.forward()</code> method are automatically removed. It must implement <code>__len__</code>.`,name:"test_dataset"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L1043"}}),Ht=new x({props:{name:"get_train_dataloader",anchor:"transformers.Trainer.get_train_dataloader",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L925"}}),Rt=new x({props:{name:"hyperparameter_search",anchor:"transformers.Trainer.hyperparameter_search",parameters:[{name:"hp_space",val:": Optional = None"},{name:"compute_objective",val:": Optional = None"},{name:"n_trials",val:": int = 20"},{name:"direction",val:": Union = 'minimize'"},{name:"backend",val:": Union = None"},{name:"hp_name",val:": Optional = None"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.Trainer.hyperparameter_search.hp_space",description:`<strong>hp_space</strong> (<code>Callable[["optuna.Trial"], Dict[str, float]]</code>, <em>optional</em>) — | |
| A function that defines the hyperparameter search space. Will default to | |
| <code>default_hp_space_optuna()</code> or <code>default_hp_space_ray()</code> or | |
| <code>default_hp_space_sigopt()</code> depending on your backend.`,name:"hp_space"},{anchor:"transformers.Trainer.hyperparameter_search.compute_objective",description:`<strong>compute_objective</strong> (<code>Callable[[Dict[str, float]], float]</code>, <em>optional</em>) — | |
| A function computing the objective to minimize or maximize from the metrics returned by the <code>evaluate</code> | |
| method. Will default to <code>default_compute_objective()</code>.`,name:"compute_objective"},{anchor:"transformers.Trainer.hyperparameter_search.n_trials",description:`<strong>n_trials</strong> (<code>int</code>, <em>optional</em>, defaults to 100) — | |
| The number of trial runs to test.`,name:"n_trials"},{anchor:"transformers.Trainer.hyperparameter_search.direction",description:`<strong>direction</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>, defaults to <code>"minimize"</code>) — | |
| If it’s single objective optimization, direction is <code>str</code>, can be <code>"minimize"</code> or <code>"maximize"</code>, you | |
| should pick <code>"minimize"</code> when optimizing the validation loss, <code>"maximize"</code> when optimizing one or | |
| several metrics. If it’s multi objectives optimization, direction is <code>List[str]</code>, can be List of | |
| <code>"minimize"</code> and <code>"maximize"</code>, you should pick <code>"minimize"</code> when optimizing the validation loss, | |
| <code>"maximize"</code> when optimizing one or several metrics.`,name:"direction"},{anchor:"transformers.Trainer.hyperparameter_search.backend",description:`<strong>backend</strong> (<code>str</code> or <code>~training_utils.HPSearchBackend</code>, <em>optional</em>) — | |
| The backend to use for hyperparameter search. Will default to optuna or Ray Tune or SigOpt, depending | |
| on which one is installed. If all are installed, will default to optuna.`,name:"backend"},{anchor:"transformers.Trainer.hyperparameter_search.hp_name",description:`<strong>hp_name</strong> (<code>Callable[["optuna.Trial"], str]]</code>, <em>optional</em>) — | |
| A function that defines the trial/run name. Will default to None.`,name:"hp_name"},{anchor:"transformers.Trainer.hyperparameter_search.kwargs",description:`<strong>kwargs</strong> (<code>Dict[str, Any]</code>, <em>optional</em>) — | |
| Additional keyword arguments passed along to <code>optuna.create_study</code> or <code>ray.tune.run</code>. For more | |
| information see:</p> | |
| <ul> | |
| <li>the documentation of | |
| <a href="https://optuna.readthedocs.io/en/stable/reference/generated/optuna.study.create_study.html" rel="nofollow">optuna.create_study</a></li> | |
| <li>the documentation of <a href="https://docs.ray.io/en/latest/tune/api_docs/execution.html#tune-run" rel="nofollow">tune.run</a></li> | |
| <li>the documentation of <a href="https://app.sigopt.com/docs/endpoints/experiments/create" rel="nofollow">sigopt</a></li> | |
| </ul>`,name:"kwargs"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L3345",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>All the information about the best run or best | |
| runs for multi-objective optimization. Experiment summary can be found in <code>run_summary</code> attribute for Ray | |
| backend.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>[<code>trainer_utils.BestRun</code> or <code>List[trainer_utils.BestRun]</code>]</p> | |
| `}}),We=new Eo({props:{warning:!0,$$slots:{default:[wm]},$$scope:{ctx:C}}}),Jt=new x({props:{name:"init_hf_repo",anchor:"transformers.Trainer.init_hf_repo",parameters:[{name:"token",val:": Optional = None"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L4362"}}),Et=new x({props:{name:"is_local_process_zero",anchor:"transformers.Trainer.is_local_process_zero",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L3597"}}),Bt=new x({props:{name:"is_world_process_zero",anchor:"transformers.Trainer.is_world_process_zero",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L3604"}}),Gt=new x({props:{name:"log",anchor:"transformers.Trainer.log",parameters:[{name:"logs",val:": Dict"}],parametersDescription:[{anchor:"transformers.Trainer.log.logs",description:`<strong>logs</strong> (<code>Dict[str, float]</code>) — | |
| The values to log.`,name:"logs"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L3424"}}),Vt=new x({props:{name:"log_metrics",anchor:"transformers.Trainer.log_metrics",parameters:[{name:"split",val:""},{name:"metrics",val:""}],parametersDescription:[{anchor:"transformers.Trainer.log_metrics.split",description:`<strong>split</strong> (<code>str</code>) — | |
| Mode/split name: one of <code>train</code>, <code>eval</code>, <code>test</code>`,name:"split"},{anchor:"transformers.Trainer.log_metrics.metrics",description:`<strong>metrics</strong> (<code>Dict[str, float]</code>) — | |
| The metrics returned from train/evaluate/predictmetrics: metrics dict`,name:"metrics"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer_pt_utils.py#L970"}}),Oe=new be({props:{anchor:"transformers.Trainer.log_metrics.example",$$slots:{default:[xm]},$$scope:{ctx:C}}}),Xt=new x({props:{name:"metrics_format",anchor:"transformers.Trainer.metrics_format",parameters:[{name:"metrics",val:": Dict"}],parametersDescription:[{anchor:"transformers.Trainer.metrics_format.metrics",description:`<strong>metrics</strong> (<code>Dict[str, float]</code>) — | |
| The metrics returned from train/evaluate/predict`,name:"metrics"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer_pt_utils.py#L944",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The reformatted metrics</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>metrics (<code>Dict[str, float]</code>)</p> | |
| `}}),Zt=new x({props:{name:"num_examples",anchor:"transformers.Trainer.num_examples",parameters:[{name:"dataloader",val:": DataLoader"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L1603"}}),Yt=new x({props:{name:"num_tokens",anchor:"transformers.Trainer.num_tokens",parameters:[{name:"train_dl",val:": DataLoader"},{name:"max_steps",val:": Optional = None"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L1617"}}),Qt=new x({props:{name:"pop_callback",anchor:"transformers.Trainer.pop_callback",parameters:[{name:"callback",val:""}],parametersDescription:[{anchor:"transformers.Trainer.pop_callback.callback",description:"<strong>callback</strong> (<code>type</code> or [`~transformers.TrainerCallback]<code>) -- A </code>TrainerCallback<code>class or an instance of a</code>TrainerCallback`. In the\nfirst case, will pop the first member of that class found in the list of callbacks.",name:"callback"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L800",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The callback removed, if found.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>TrainerCallback</code></p> | |
| `}}),Kt=new x({props:{name:"predict",anchor:"transformers.Trainer.predict",parameters:[{name:"test_dataset",val:": Dataset"},{name:"ignore_keys",val:": Optional = None"},{name:"metric_key_prefix",val:": str = 'test'"}],parametersDescription:[{anchor:"transformers.Trainer.predict.test_dataset",description:`<strong>test_dataset</strong> (<code>Dataset</code>) — | |
| Dataset to run the predictions on. If it is an <code>datasets.Dataset</code>, columns not accepted by the | |
| <code>model.forward()</code> method are automatically removed. Has to implement the method <code>__len__</code>`,name:"test_dataset"},{anchor:"transformers.Trainer.predict.ignore_keys",description:`<strong>ignore_keys</strong> (<code>List[str]</code>, <em>optional</em>) — | |
| A list of keys in the output of your model (if it is a dictionary) that should be ignored when | |
| gathering predictions.`,name:"ignore_keys"},{anchor:"transformers.Trainer.predict.metric_key_prefix",description:`<strong>metric_key_prefix</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"test"</code>) — | |
| An optional prefix to be used as the metrics key prefix. For example the metrics “bleu” will be named | |
| “test_bleu” if the prefix is “test” (default)`,name:"metric_key_prefix"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L3942"}}),Ee=new Eo({props:{$$slots:{default:[qm]},$$scope:{ctx:C}}}),eo=new x({props:{name:"prediction_loop",anchor:"transformers.Trainer.prediction_loop",parameters:[{name:"dataloader",val:": DataLoader"},{name:"description",val:": str"},{name:"prediction_loss_only",val:": Optional = None"},{name:"ignore_keys",val:": Optional = None"},{name:"metric_key_prefix",val:": str = 'eval'"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L4607"}}),to=new x({props:{name:"prediction_step",anchor:"transformers.Trainer.prediction_step",parameters:[{name:"model",val:": Module"},{name:"inputs",val:": Dict"},{name:"prediction_loss_only",val:": bool"},{name:"ignore_keys",val:": Optional = None"}],parametersDescription:[{anchor:"transformers.Trainer.prediction_step.model",description:`<strong>model</strong> (<code>nn.Module</code>) — | |
| The model to evaluate.`,name:"model"},{anchor:"transformers.Trainer.prediction_step.inputs",description:`<strong>inputs</strong> (<code>Dict[str, Union[torch.Tensor, Any]]</code>) — | |
| The inputs and targets of the model.</p> | |
| <p>The dictionary will be unpacked before being fed to the model. Most models expect the targets under the | |
| argument <code>labels</code>. Check your model’s documentation for all accepted arguments.`,name:"inputs"},{anchor:"transformers.Trainer.prediction_step.prediction_loss_only",description:`<strong>prediction_loss_only</strong> (<code>bool</code>) — | |
| Whether or not to return the loss only.`,name:"prediction_loss_only"},{anchor:"transformers.Trainer.prediction_step.ignore_keys",description:`<strong>ignore_keys</strong> (<code>List[str]</code>, <em>optional</em>) — | |
| A list of keys in the output of your model (if it is a dictionary) that should be ignored when | |
| gathering predictions.`,name:"ignore_keys"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L4239",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A tuple with the loss, | |
| logits and labels (each being optional).</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>Tuple[Optional[torch.Tensor], Optional[torch.Tensor], Optional[torch.Tensor]]</p> | |
| `}}),oo=new x({props:{name:"propagate_args_to_deepspeed",anchor:"transformers.Trainer.propagate_args_to_deepspeed",parameters:[{name:"auto_find_batch_size",val:" = False"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L4943"}}),no=new x({props:{name:"push_to_hub",anchor:"transformers.Trainer.push_to_hub",parameters:[{name:"commit_message",val:": Optional = 'End of training'"},{name:"blocking",val:": bool = True"},{name:"token",val:": Optional = None"},{name:"revision",val:": Optional = None"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.Trainer.push_to_hub.commit_message",description:`<strong>commit_message</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"End of training"</code>) — | |
| Message to commit while pushing.`,name:"commit_message"},{anchor:"transformers.Trainer.push_to_hub.blocking",description:`<strong>blocking</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether the function should return only when the <code>git push</code> has finished.`,name:"blocking"},{anchor:"transformers.Trainer.push_to_hub.token",description:`<strong>token</strong> (<code>str</code>, <em>optional</em>, defaults to <code>None</code>) — | |
| Token with write permission to overwrite Trainer’s original args.`,name:"token"},{anchor:"transformers.Trainer.push_to_hub.revision",description:`<strong>revision</strong> (<code>str</code>, <em>optional</em>) — | |
| The git revision to commit from. Defaults to the head of the “main” branch.`,name:"revision"},{anchor:"transformers.Trainer.push_to_hub.kwargs",description:`<strong>kwargs</strong> (<code>Dict[str, Any]</code>, <em>optional</em>) — | |
| Additional keyword arguments passed along to <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer.create_model_card">create_model_card()</a>.`,name:"kwargs"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L4527",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The URL of the repository where the model was pushed if <code>blocking=False</code>, or a <code>Future</code> object tracking the | |
| progress of the commit if <code>blocking=True</code>.</p> | |
| `}}),ro=new x({props:{name:"remove_callback",anchor:"transformers.Trainer.remove_callback",parameters:[{name:"callback",val:""}],parametersDescription:[{anchor:"transformers.Trainer.remove_callback.callback",description:"<strong>callback</strong> (<code>type</code> or [`~transformers.TrainerCallback]<code>) -- A </code>TrainerCallback<code>class or an instance of a</code>TrainerCallback`. In the\nfirst case, will remove the first member of that class found in the list of callbacks.",name:"callback"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L816"}}),ao=new x({props:{name:"save_metrics",anchor:"transformers.Trainer.save_metrics",parameters:[{name:"split",val:""},{name:"metrics",val:""},{name:"combined",val:" = True"}],parametersDescription:[{anchor:"transformers.Trainer.save_metrics.split",description:`<strong>split</strong> (<code>str</code>) — | |
| Mode/split name: one of <code>train</code>, <code>eval</code>, <code>test</code>, <code>all</code>`,name:"split"},{anchor:"transformers.Trainer.save_metrics.metrics",description:`<strong>metrics</strong> (<code>Dict[str, float]</code>) — | |
| The metrics returned from train/evaluate/predict`,name:"metrics"},{anchor:"transformers.Trainer.save_metrics.combined",description:`<strong>combined</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Creates combined metrics by updating <code>all_results.json</code> with metrics of this call`,name:"combined"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer_pt_utils.py#L1060"}}),so=new x({props:{name:"save_model",anchor:"transformers.Trainer.save_model",parameters:[{name:"output_dir",val:": Optional = None"},{name:"_internal_call",val:": bool = False"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L3616"}}),io=new x({props:{name:"save_state",anchor:"transformers.Trainer.save_state",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer_pt_utils.py#L1098"}}),lo=new x({props:{name:"train",anchor:"transformers.Trainer.train",parameters:[{name:"resume_from_checkpoint",val:": Union = None"},{name:"trial",val:": Union = None"},{name:"ignore_keys_for_eval",val:": Optional = None"},{name:"**kwargs",val:""}],parametersDescription:[{anchor:"transformers.Trainer.train.resume_from_checkpoint",description:`<strong>resume_from_checkpoint</strong> (<code>str</code> or <code>bool</code>, <em>optional</em>) — | |
| If a <code>str</code>, local path to a saved checkpoint as saved by a previous instance of <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>. If a | |
| <code>bool</code> and equals <code>True</code>, load the last checkpoint in <em>args.output_dir</em> as saved by a previous instance | |
| of <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>. If present, training will resume from the model/optimizer/scheduler states loaded here.`,name:"resume_from_checkpoint"},{anchor:"transformers.Trainer.train.trial",description:`<strong>trial</strong> (<code>optuna.Trial</code> or <code>Dict[str, Any]</code>, <em>optional</em>) — | |
| The trial run or the hyperparameter dictionary for hyperparameter search.`,name:"trial"},{anchor:"transformers.Trainer.train.ignore_keys_for_eval",description:`<strong>ignore_keys_for_eval</strong> (<code>List[str]</code>, <em>optional</em>) — | |
| A list of keys in the output of your model (if it is a dictionary) that should be ignored when | |
| gathering predictions for evaluation during the training.`,name:"ignore_keys_for_eval"},{anchor:"transformers.Trainer.train.kwargs",description:`<strong>kwargs</strong> (<code>Dict[str, Any]</code>, <em>optional</em>) — | |
| Additional keyword arguments used to hide deprecated arguments`,name:"kwargs"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L1983"}}),co=new x({props:{name:"training_step",anchor:"transformers.Trainer.training_step",parameters:[{name:"model",val:": Module"},{name:"inputs",val:": Dict"}],parametersDescription:[{anchor:"transformers.Trainer.training_step.model",description:`<strong>model</strong> (<code>nn.Module</code>) — | |
| The model to train.`,name:"model"},{anchor:"transformers.Trainer.training_step.inputs",description:`<strong>inputs</strong> (<code>Dict[str, Union[torch.Tensor, Any]]</code>) — | |
| The inputs and targets of the model.</p> | |
| <p>The dictionary will be unpacked before being fed to the model. Most models expect the targets under the | |
| argument <code>labels</code>. Check your model’s documentation for all accepted arguments.`,name:"inputs"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer.py#L3495",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>The tensor with training loss on this batch.</p> | |
| `,returnType:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p><code>torch.Tensor</code></p> | |
| `}}),mo=new ua({props:{title:"Seq2SeqTrainer",local:"transformers.Seq2SeqTrainer ][ transformers.Seq2SeqTrainer",headingTag:"h2"}}),po=new x({props:{name:"class transformers.Seq2SeqTrainer",anchor:"transformers.Seq2SeqTrainer",parameters:[{name:"model",val:": Union = None"},{name:"args",val:": TrainingArguments = None"},{name:"data_collator",val:": Optional = None"},{name:"train_dataset",val:": Union = None"},{name:"eval_dataset",val:": Union = None"},{name:"processing_class",val:": Union = None"},{name:"model_init",val:": Optional = None"},{name:"compute_metrics",val:": Optional = None"},{name:"callbacks",val:": Optional = None"},{name:"optimizers",val:": Tuple = (None, None)"},{name:"preprocess_logits_for_metrics",val:": Optional = None"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer_seq2seq.py#L51"}}),uo=new x({props:{name:"evaluate",anchor:"transformers.Seq2SeqTrainer.evaluate",parameters:[{name:"eval_dataset",val:": Optional = None"},{name:"ignore_keys",val:": Optional = None"},{name:"metric_key_prefix",val:": str = 'eval'"},{name:"**gen_kwargs",val:""}],parametersDescription:[{anchor:"transformers.Seq2SeqTrainer.evaluate.eval_dataset",description:`<strong>eval_dataset</strong> (<code>Dataset</code>, <em>optional</em>) — | |
| Pass a dataset if you wish to override <code>self.eval_dataset</code>. If it is an <code>Dataset</code>, columns | |
| not accepted by the <code>model.forward()</code> method are automatically removed. It must implement the <code>__len__</code> | |
| method.`,name:"eval_dataset"},{anchor:"transformers.Seq2SeqTrainer.evaluate.ignore_keys",description:`<strong>ignore_keys</strong> (<code>List[str]</code>, <em>optional</em>) — | |
| A list of keys in the output of your model (if it is a dictionary) that should be ignored when | |
| gathering predictions.`,name:"ignore_keys"},{anchor:"transformers.Seq2SeqTrainer.evaluate.metric_key_prefix",description:`<strong>metric_key_prefix</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"eval"</code>) — | |
| An optional prefix to be used as the metrics key prefix. For example the metrics “bleu” will be named | |
| “eval_bleu” if the prefix is <code>"eval"</code> (default)`,name:"metric_key_prefix"},{anchor:"transformers.Seq2SeqTrainer.evaluate.max_length",description:`<strong>max_length</strong> (<code>int</code>, <em>optional</em>) — | |
| The maximum target length to use when predicting with the generate method.`,name:"max_length"},{anchor:"transformers.Seq2SeqTrainer.evaluate.num_beams",description:`<strong>num_beams</strong> (<code>int</code>, <em>optional</em>) — | |
| Number of beams for beam search that will be used when predicting with the generate method. 1 means no | |
| beam search. | |
| gen_kwargs — | |
| Additional <code>generate</code> specific kwargs.`,name:"num_beams"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer_seq2seq.py#L138",returnDescription:`<script context="module">export const metadata = 'undefined';<\/script> | |
| <p>A dictionary containing the evaluation loss and the potential metrics computed from the predictions. The | |
| dictionary also contains the epoch number which comes from the training state.</p> | |
| `}}),go=new x({props:{name:"predict",anchor:"transformers.Seq2SeqTrainer.predict",parameters:[{name:"test_dataset",val:": Dataset"},{name:"ignore_keys",val:": Optional = None"},{name:"metric_key_prefix",val:": str = 'test'"},{name:"**gen_kwargs",val:""}],parametersDescription:[{anchor:"transformers.Seq2SeqTrainer.predict.test_dataset",description:`<strong>test_dataset</strong> (<code>Dataset</code>) — | |
| Dataset to run the predictions on. If it is a <code>Dataset</code>, columns not accepted by the | |
| <code>model.forward()</code> method are automatically removed. Has to implement the method <code>__len__</code>`,name:"test_dataset"},{anchor:"transformers.Seq2SeqTrainer.predict.ignore_keys",description:`<strong>ignore_keys</strong> (<code>List[str]</code>, <em>optional</em>) — | |
| A list of keys in the output of your model (if it is a dictionary) that should be ignored when | |
| gathering predictions.`,name:"ignore_keys"},{anchor:"transformers.Seq2SeqTrainer.predict.metric_key_prefix",description:`<strong>metric_key_prefix</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"eval"</code>) — | |
| An optional prefix to be used as the metrics key prefix. For example the metrics “bleu” will be named | |
| “eval_bleu” if the prefix is <code>"eval"</code> (default)`,name:"metric_key_prefix"},{anchor:"transformers.Seq2SeqTrainer.predict.max_length",description:`<strong>max_length</strong> (<code>int</code>, <em>optional</em>) — | |
| The maximum target length to use when predicting with the generate method.`,name:"max_length"},{anchor:"transformers.Seq2SeqTrainer.predict.num_beams",description:`<strong>num_beams</strong> (<code>int</code>, <em>optional</em>) — | |
| Number of beams for beam search that will be used when predicting with the generate method. 1 means no | |
| beam search. | |
| gen_kwargs — | |
| Additional <code>generate</code> specific kwargs.`,name:"num_beams"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/trainer_seq2seq.py#L194"}}),Ze=new Eo({props:{$$slots:{default:[km]},$$scope:{ctx:C}}}),ho=new ua({props:{title:"TrainingArguments",local:"transformers.TrainingArguments ][ transformers.TrainingArguments",headingTag:"h2"}}),fo=new x({props:{name:"class transformers.TrainingArguments",anchor:"transformers.TrainingArguments",parameters:[{name:"output_dir",val:": str"},{name:"overwrite_output_dir",val:": bool = False"},{name:"do_train",val:": bool = False"},{name:"do_eval",val:": bool = False"},{name:"do_predict",val:": bool = False"},{name:"eval_strategy",val:": Union = 'no'"},{name:"prediction_loss_only",val:": bool = False"},{name:"per_device_train_batch_size",val:": int = 8"},{name:"per_device_eval_batch_size",val:": int = 8"},{name:"per_gpu_train_batch_size",val:": Optional = None"},{name:"per_gpu_eval_batch_size",val:": Optional = None"},{name:"gradient_accumulation_steps",val:": int = 1"},{name:"eval_accumulation_steps",val:": Optional = None"},{name:"eval_delay",val:": Optional = 0"},{name:"torch_empty_cache_steps",val:": Optional = None"},{name:"learning_rate",val:": float = 5e-05"},{name:"weight_decay",val:": float = 0.0"},{name:"adam_beta1",val:": float = 0.9"},{name:"adam_beta2",val:": float = 0.999"},{name:"adam_epsilon",val:": float = 1e-08"},{name:"max_grad_norm",val:": float = 1.0"},{name:"num_train_epochs",val:": float = 3.0"},{name:"max_steps",val:": int = -1"},{name:"lr_scheduler_type",val:": Union = 'linear'"},{name:"lr_scheduler_kwargs",val:": Union = <factory>"},{name:"warmup_ratio",val:": float = 0.0"},{name:"warmup_steps",val:": int = 0"},{name:"log_level",val:": Optional = 'passive'"},{name:"log_level_replica",val:": Optional = 'warning'"},{name:"log_on_each_node",val:": bool = True"},{name:"logging_dir",val:": Optional = None"},{name:"logging_strategy",val:": Union = 'steps'"},{name:"logging_first_step",val:": bool = False"},{name:"logging_steps",val:": float = 500"},{name:"logging_nan_inf_filter",val:": bool = True"},{name:"save_strategy",val:": Union = 'steps'"},{name:"save_steps",val:": float = 500"},{name:"save_total_limit",val:": Optional = None"},{name:"save_safetensors",val:": Optional = True"},{name:"save_on_each_node",val:": bool = False"},{name:"save_only_model",val:": bool = False"},{name:"restore_callback_states_from_checkpoint",val:": bool = False"},{name:"no_cuda",val:": bool = False"},{name:"use_cpu",val:": bool = False"},{name:"use_mps_device",val:": bool = False"},{name:"seed",val:": int = 42"},{name:"data_seed",val:": Optional = None"},{name:"jit_mode_eval",val:": bool = False"},{name:"use_ipex",val:": bool = False"},{name:"bf16",val:": bool = False"},{name:"fp16",val:": bool = False"},{name:"fp16_opt_level",val:": str = 'O1'"},{name:"half_precision_backend",val:": str = 'auto'"},{name:"bf16_full_eval",val:": bool = False"},{name:"fp16_full_eval",val:": bool = False"},{name:"tf32",val:": Optional = None"},{name:"local_rank",val:": int = -1"},{name:"ddp_backend",val:": Optional = None"},{name:"tpu_num_cores",val:": Optional = None"},{name:"tpu_metrics_debug",val:": bool = False"},{name:"debug",val:": Union = ''"},{name:"dataloader_drop_last",val:": bool = False"},{name:"eval_steps",val:": Optional = None"},{name:"dataloader_num_workers",val:": int = 0"},{name:"dataloader_prefetch_factor",val:": Optional = None"},{name:"past_index",val:": int = -1"},{name:"run_name",val:": Optional = None"},{name:"disable_tqdm",val:": Optional = None"},{name:"remove_unused_columns",val:": Optional = True"},{name:"label_names",val:": Optional = None"},{name:"load_best_model_at_end",val:": Optional = False"},{name:"metric_for_best_model",val:": Optional = None"},{name:"greater_is_better",val:": Optional = None"},{name:"ignore_data_skip",val:": bool = False"},{name:"fsdp",val:": Union = ''"},{name:"fsdp_min_num_params",val:": int = 0"},{name:"fsdp_config",val:": Union = None"},{name:"fsdp_transformer_layer_cls_to_wrap",val:": Optional = None"},{name:"accelerator_config",val:": Union = None"},{name:"deepspeed",val:": Union = None"},{name:"label_smoothing_factor",val:": float = 0.0"},{name:"optim",val:": Union = 'adamw_torch'"},{name:"optim_args",val:": Optional = None"},{name:"adafactor",val:": bool = False"},{name:"group_by_length",val:": bool = False"},{name:"length_column_name",val:": Optional = 'length'"},{name:"report_to",val:": Union = None"},{name:"ddp_find_unused_parameters",val:": Optional = None"},{name:"ddp_bucket_cap_mb",val:": Optional = None"},{name:"ddp_broadcast_buffers",val:": Optional = None"},{name:"dataloader_pin_memory",val:": bool = True"},{name:"dataloader_persistent_workers",val:": bool = False"},{name:"skip_memory_metrics",val:": bool = True"},{name:"use_legacy_prediction_loop",val:": bool = False"},{name:"push_to_hub",val:": bool = False"},{name:"resume_from_checkpoint",val:": Optional = None"},{name:"hub_model_id",val:": Optional = None"},{name:"hub_strategy",val:": Union = 'every_save'"},{name:"hub_token",val:": Optional = None"},{name:"hub_private_repo",val:": bool = False"},{name:"hub_always_push",val:": bool = False"},{name:"gradient_checkpointing",val:": bool = False"},{name:"gradient_checkpointing_kwargs",val:": Union = None"},{name:"include_inputs_for_metrics",val:": bool = False"},{name:"include_for_metrics",val:": List = <factory>"},{name:"eval_do_concat_batches",val:": bool = True"},{name:"fp16_backend",val:": str = 'auto'"},{name:"evaluation_strategy",val:": Union = None"},{name:"push_to_hub_model_id",val:": Optional = None"},{name:"push_to_hub_organization",val:": Optional = None"},{name:"push_to_hub_token",val:": Optional = None"},{name:"mp_parameters",val:": str = ''"},{name:"auto_find_batch_size",val:": bool = False"},{name:"full_determinism",val:": bool = False"},{name:"torchdynamo",val:": Optional = None"},{name:"ray_scope",val:": Optional = 'last'"},{name:"ddp_timeout",val:": Optional = 1800"},{name:"torch_compile",val:": bool = False"},{name:"torch_compile_backend",val:": Optional = None"},{name:"torch_compile_mode",val:": Optional = None"},{name:"dispatch_batches",val:": Optional = None"},{name:"split_batches",val:": Optional = None"},{name:"include_tokens_per_second",val:": Optional = False"},{name:"include_num_input_tokens_seen",val:": Optional = False"},{name:"neftune_noise_alpha",val:": Optional = None"},{name:"optim_target_modules",val:": Union = None"},{name:"batch_eval_metrics",val:": bool = False"},{name:"eval_on_start",val:": bool = False"},{name:"use_liger_kernel",val:": Optional = False"},{name:"eval_use_gather_object",val:": Optional = False"}],parametersDescription:[{anchor:"transformers.TrainingArguments.output_dir",description:`<strong>output_dir</strong> (<code>str</code>) — | |
| The output directory where the model predictions and checkpoints will be written.`,name:"output_dir"},{anchor:"transformers.TrainingArguments.overwrite_output_dir",description:`<strong>overwrite_output_dir</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| If <code>True</code>, overwrite the content of the output directory. Use this to continue training if <code>output_dir</code> | |
| points to a checkpoint directory.`,name:"overwrite_output_dir"},{anchor:"transformers.TrainingArguments.do_train",description:`<strong>do_train</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to run training or not. This argument is not directly used by <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>, it’s intended to be used | |
| by your training/evaluation scripts instead. See the <a href="https://github.com/huggingface/transformers/tree/main/examples" rel="nofollow">example | |
| scripts</a> for more details.`,name:"do_train"},{anchor:"transformers.TrainingArguments.do_eval",description:`<strong>do_eval</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether to run evaluation on the validation set or not. Will be set to <code>True</code> if <code>eval_strategy</code> is | |
| different from <code>"no"</code>. This argument is not directly used by <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>, it’s intended to be used by your | |
| training/evaluation scripts instead. See the <a href="https://github.com/huggingface/transformers/tree/main/examples" rel="nofollow">example | |
| scripts</a> for more details.`,name:"do_eval"},{anchor:"transformers.TrainingArguments.do_predict",description:`<strong>do_predict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to run predictions on the test set or not. This argument is not directly used by <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>, it’s | |
| intended to be used by your training/evaluation scripts instead. See the <a href="https://github.com/huggingface/transformers/tree/main/examples" rel="nofollow">example | |
| scripts</a> for more details.`,name:"do_predict"},{anchor:"transformers.TrainingArguments.eval_strategy",description:`<strong>eval_strategy</strong> (<code>str</code> or <code>IntervalStrategy</code>, <em>optional</em>, defaults to <code>"no"</code>) — | |
| The evaluation strategy to adopt during training. Possible values are:</p> | |
| <ul> | |
| <li><code>"no"</code>: No evaluation is done during training.</li> | |
| <li><code>"steps"</code>: Evaluation is done (and logged) every <code>eval_steps</code>.</li> | |
| <li><code>"epoch"</code>: Evaluation is done at the end of each epoch.</li> | |
| </ul>`,name:"eval_strategy"},{anchor:"transformers.TrainingArguments.prediction_loss_only",description:`<strong>prediction_loss_only</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| When performing evaluation and generating predictions, only returns the loss.`,name:"prediction_loss_only"},{anchor:"transformers.TrainingArguments.per_device_train_batch_size",description:`<strong>per_device_train_batch_size</strong> (<code>int</code>, <em>optional</em>, defaults to 8) — | |
| The batch size per GPU/XPU/TPU/MPS/NPU core/CPU for training.`,name:"per_device_train_batch_size"},{anchor:"transformers.TrainingArguments.per_device_eval_batch_size",description:`<strong>per_device_eval_batch_size</strong> (<code>int</code>, <em>optional</em>, defaults to 8) — | |
| The batch size per GPU/XPU/TPU/MPS/NPU core/CPU for evaluation.`,name:"per_device_eval_batch_size"},{anchor:"transformers.TrainingArguments.gradient_accumulation_steps",description:`<strong>gradient_accumulation_steps</strong> (<code>int</code>, <em>optional</em>, defaults to 1) — | |
| Number of updates steps to accumulate the gradients for, before performing a backward/update pass.</p> | |
| <div class="course-tip course-tip-orange bg-gradient-to-br dark:bg-gradient-to-r before:border-orange-500 dark:before:border-orange-800 from-orange-50 dark:from-gray-900 to-white dark:to-gray-950 border border-orange-50 text-orange-700 dark:text-gray-400"> | |
| <p>When using gradient accumulation, one step is counted as one step with backward pass. Therefore, logging, | |
| evaluation, save will be conducted every <code>gradient_accumulation_steps * xxx_step</code> training examples.</p> | |
| </div>`,name:"gradient_accumulation_steps"},{anchor:"transformers.TrainingArguments.eval_accumulation_steps",description:`<strong>eval_accumulation_steps</strong> (<code>int</code>, <em>optional</em>) — | |
| Number of predictions steps to accumulate the output tensors for, before moving the results to the CPU. If | |
| left unset, the whole predictions are accumulated on GPU/NPU/TPU before being moved to the CPU (faster but | |
| requires more memory).`,name:"eval_accumulation_steps"},{anchor:"transformers.TrainingArguments.eval_delay",description:`<strong>eval_delay</strong> (<code>float</code>, <em>optional</em>) — | |
| Number of epochs or steps to wait for before the first evaluation can be performed, depending on the | |
| eval_strategy.`,name:"eval_delay"},{anchor:"transformers.TrainingArguments.torch_empty_cache_steps",description:`<strong>torch_empty_cache_steps</strong> (<code>int</code>, <em>optional</em>) — | |
| Number of steps to wait before calling <code>torch.<device>.empty_cache()</code>. If left unset or set to None, cache will not be emptied.</p> | |
| <div class="course-tip bg-gradient-to-br dark:bg-gradient-to-r before:border-green-500 dark:before:border-green-800 from-green-50 dark:from-gray-900 to-white dark:to-gray-950 border border-green-50 text-green-700 dark:text-gray-400"> | |
| <p>This can help avoid CUDA out-of-memory errors by lowering peak VRAM usage at a cost of about <a href="https://github.com/huggingface/transformers/issues/31372" rel="nofollow">10% slower performance</a>.</p> | |
| </div>`,name:"torch_empty_cache_steps"},{anchor:"transformers.TrainingArguments.learning_rate",description:`<strong>learning_rate</strong> (<code>float</code>, <em>optional</em>, defaults to 5e-5) — | |
| The initial learning rate for <code>AdamW</code> optimizer.`,name:"learning_rate"},{anchor:"transformers.TrainingArguments.weight_decay",description:`<strong>weight_decay</strong> (<code>float</code>, <em>optional</em>, defaults to 0) — | |
| The weight decay to apply (if not zero) to all layers except all bias and LayerNorm weights in <code>AdamW</code> | |
| optimizer.`,name:"weight_decay"},{anchor:"transformers.TrainingArguments.adam_beta1",description:`<strong>adam_beta1</strong> (<code>float</code>, <em>optional</em>, defaults to 0.9) — | |
| The beta1 hyperparameter for the <code>AdamW</code> optimizer.`,name:"adam_beta1"},{anchor:"transformers.TrainingArguments.adam_beta2",description:`<strong>adam_beta2</strong> (<code>float</code>, <em>optional</em>, defaults to 0.999) — | |
| The beta2 hyperparameter for the <code>AdamW</code> optimizer.`,name:"adam_beta2"},{anchor:"transformers.TrainingArguments.adam_epsilon",description:`<strong>adam_epsilon</strong> (<code>float</code>, <em>optional</em>, defaults to 1e-8) — | |
| The epsilon hyperparameter for the <code>AdamW</code> optimizer.`,name:"adam_epsilon"},{anchor:"transformers.TrainingArguments.max_grad_norm",description:`<strong>max_grad_norm</strong> (<code>float</code>, <em>optional</em>, defaults to 1.0) — | |
| Maximum gradient norm (for gradient clipping).`,name:"max_grad_norm"},{anchor:"transformers.TrainingArguments.num_train_epochs(float,",description:`<strong>num_train_epochs(<code>float</code>,</strong> <em>optional</em>, defaults to 3.0) — | |
| Total number of training epochs to perform (if not an integer, will perform the decimal part percents of | |
| the last epoch before stopping training).`,name:"num_train_epochs(float,"},{anchor:"transformers.TrainingArguments.max_steps",description:`<strong>max_steps</strong> (<code>int</code>, <em>optional</em>, defaults to -1) — | |
| If set to a positive number, the total number of training steps to perform. Overrides <code>num_train_epochs</code>. | |
| For a finite dataset, training is reiterated through the dataset (if all data is exhausted) until | |
| <code>max_steps</code> is reached.`,name:"max_steps"},{anchor:"transformers.TrainingArguments.lr_scheduler_type",description:`<strong>lr_scheduler_type</strong> (<code>str</code> or <code>SchedulerType</code>, <em>optional</em>, defaults to <code>"linear"</code>) — | |
| The scheduler type to use. See the documentation of <code>SchedulerType</code> for all possible values.`,name:"lr_scheduler_type"},{anchor:"transformers.TrainingArguments.lr_scheduler_kwargs",description:`<strong>lr_scheduler_kwargs</strong> (‘dict’, <em>optional</em>, defaults to {}) — | |
| The extra arguments for the lr_scheduler. See the documentation of each scheduler for possible values.`,name:"lr_scheduler_kwargs"},{anchor:"transformers.TrainingArguments.warmup_ratio",description:`<strong>warmup_ratio</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| Ratio of total training steps used for a linear warmup from 0 to <code>learning_rate</code>.`,name:"warmup_ratio"},{anchor:"transformers.TrainingArguments.warmup_steps",description:`<strong>warmup_steps</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| Number of steps used for a linear warmup from 0 to <code>learning_rate</code>. Overrides any effect of <code>warmup_ratio</code>.`,name:"warmup_steps"},{anchor:"transformers.TrainingArguments.log_level",description:`<strong>log_level</strong> (<code>str</code>, <em>optional</em>, defaults to <code>passive</code>) — | |
| Logger log level to use on the main process. Possible choices are the log levels as strings: ‘debug’, | |
| ‘info’, ‘warning’, ‘error’ and ‘critical’, plus a ‘passive’ level which doesn’t set anything and keeps the | |
| current log level for the Transformers library (which will be <code>"warning"</code> by default).`,name:"log_level"},{anchor:"transformers.TrainingArguments.log_level_replica",description:`<strong>log_level_replica</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"warning"</code>) — | |
| Logger log level to use on replicas. Same choices as <code>log_level</code>”`,name:"log_level_replica"},{anchor:"transformers.TrainingArguments.log_on_each_node",description:`<strong>log_on_each_node</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| In multinode distributed training, whether to log using <code>log_level</code> once per node, or only on the main | |
| node.`,name:"log_on_each_node"},{anchor:"transformers.TrainingArguments.logging_dir",description:`<strong>logging_dir</strong> (<code>str</code>, <em>optional</em>) — | |
| <a href="https://www.tensorflow.org/tensorboard" rel="nofollow">TensorBoard</a> log directory. Will default to | |
| *output_dir/runs/<strong>CURRENT_DATETIME_HOSTNAME*</strong>.`,name:"logging_dir"},{anchor:"transformers.TrainingArguments.logging_strategy",description:`<strong>logging_strategy</strong> (<code>str</code> or <code>IntervalStrategy</code>, <em>optional</em>, defaults to <code>"steps"</code>) — | |
| The logging strategy to adopt during training. Possible values are:</p> | |
| <ul> | |
| <li><code>"no"</code>: No logging is done during training.</li> | |
| <li><code>"epoch"</code>: Logging is done at the end of each epoch.</li> | |
| <li><code>"steps"</code>: Logging is done every <code>logging_steps</code>.</li> | |
| </ul>`,name:"logging_strategy"},{anchor:"transformers.TrainingArguments.logging_first_step",description:`<strong>logging_first_step</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to log the first <code>global_step</code> or not.`,name:"logging_first_step"},{anchor:"transformers.TrainingArguments.logging_steps",description:`<strong>logging_steps</strong> (<code>int</code> or <code>float</code>, <em>optional</em>, defaults to 500) — | |
| Number of update steps between two logs if <code>logging_strategy="steps"</code>. Should be an integer or a float in | |
| range <code>[0,1)</code>. If smaller than 1, will be interpreted as ratio of total training steps.`,name:"logging_steps"},{anchor:"transformers.TrainingArguments.logging_nan_inf_filter",description:`<strong>logging_nan_inf_filter</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to filter <code>nan</code> and <code>inf</code> losses for logging. If set to <code>True</code> the loss of every step that is <code>nan</code> | |
| or <code>inf</code> is filtered and the average loss of the current logging window is taken instead.</p> | |
| <div class="course-tip bg-gradient-to-br dark:bg-gradient-to-r before:border-green-500 dark:before:border-green-800 from-green-50 dark:from-gray-900 to-white dark:to-gray-950 border border-green-50 text-green-700 dark:text-gray-400"> | |
| <p><code>logging_nan_inf_filter</code> only influences the logging of loss values, it does not change the behavior the | |
| gradient is computed or applied to the model.</p> | |
| </div>`,name:"logging_nan_inf_filter"},{anchor:"transformers.TrainingArguments.save_strategy",description:`<strong>save_strategy</strong> (<code>str</code> or <code>IntervalStrategy</code>, <em>optional</em>, defaults to <code>"steps"</code>) — | |
| The checkpoint save strategy to adopt during training. Possible values are:</p> | |
| <ul> | |
| <li><code>"no"</code>: No save is done during training.</li> | |
| <li><code>"epoch"</code>: Save is done at the end of each epoch.</li> | |
| <li><code>"steps"</code>: Save is done every <code>save_steps</code>.</li> | |
| </ul> | |
| <p>If <code>"epoch"</code> or <code>"steps"</code> is chosen, saving will also be performed at the | |
| very end of training, always.`,name:"save_strategy"},{anchor:"transformers.TrainingArguments.save_steps",description:`<strong>save_steps</strong> (<code>int</code> or <code>float</code>, <em>optional</em>, defaults to 500) — | |
| Number of updates steps before two checkpoint saves if <code>save_strategy="steps"</code>. Should be an integer or a | |
| float in range <code>[0,1)</code>. If smaller than 1, will be interpreted as ratio of total training steps.`,name:"save_steps"},{anchor:"transformers.TrainingArguments.save_total_limit",description:`<strong>save_total_limit</strong> (<code>int</code>, <em>optional</em>) — | |
| If a value is passed, will limit the total amount of checkpoints. Deletes the older checkpoints in | |
| <code>output_dir</code>. When <code>load_best_model_at_end</code> is enabled, the “best” checkpoint according to | |
| <code>metric_for_best_model</code> will always be retained in addition to the most recent ones. For example, for | |
| <code>save_total_limit=5</code> and <code>load_best_model_at_end</code>, the four last checkpoints will always be retained | |
| alongside the best model. When <code>save_total_limit=1</code> and <code>load_best_model_at_end</code>, it is possible that two | |
| checkpoints are saved: the last one and the best one (if they are different).`,name:"save_total_limit"},{anchor:"transformers.TrainingArguments.save_safetensors",description:`<strong>save_safetensors</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Use <a href="https://huggingface.co/docs/safetensors" rel="nofollow">safetensors</a> saving and loading for state dicts instead of | |
| default <code>torch.load</code> and <code>torch.save</code>.`,name:"save_safetensors"},{anchor:"transformers.TrainingArguments.save_on_each_node",description:`<strong>save_on_each_node</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| When doing multi-node distributed training, whether to save models and checkpoints on each node, or only on | |
| the main one.</p> | |
| <p>This should not be activated when the different nodes use the same storage as the files will be saved with | |
| the same names for each node.`,name:"save_on_each_node"},{anchor:"transformers.TrainingArguments.save_only_model",description:`<strong>save_only_model</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| When checkpointing, whether to only save the model, or also the optimizer, scheduler & rng state. | |
| Note that when this is true, you won’t be able to resume training from checkpoint. | |
| This enables you to save storage by not storing the optimizer, scheduler & rng state. | |
| You can only load the model using <code>from_pretrained</code> with this option set to <code>True</code>.`,name:"save_only_model"},{anchor:"transformers.TrainingArguments.restore_callback_states_from_checkpoint",description:`<strong>restore_callback_states_from_checkpoint</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to restore the callback states from the checkpoint. If <code>True</code>, will override | |
| callbacks passed to the <code>Trainer</code> if they exist in the checkpoint.”`,name:"restore_callback_states_from_checkpoint"},{anchor:"transformers.TrainingArguments.use_cpu",description:`<strong>use_cpu</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to use cpu. If set to False, we will use cuda or mps device if available.`,name:"use_cpu"},{anchor:"transformers.TrainingArguments.seed",description:`<strong>seed</strong> (<code>int</code>, <em>optional</em>, defaults to 42) — | |
| Random seed that will be set at the beginning of training. To ensure reproducibility across runs, use the | |
| <code>~Trainer.model_init</code> function to instantiate the model if it has some randomly initialized parameters.`,name:"seed"},{anchor:"transformers.TrainingArguments.data_seed",description:`<strong>data_seed</strong> (<code>int</code>, <em>optional</em>) — | |
| Random seed to be used with data samplers. If not set, random generators for data sampling will use the | |
| same seed as <code>seed</code>. This can be used to ensure reproducibility of data sampling, independent of the model | |
| seed.`,name:"data_seed"},{anchor:"transformers.TrainingArguments.jit_mode_eval",description:`<strong>jit_mode_eval</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to use PyTorch jit trace for inference.`,name:"jit_mode_eval"},{anchor:"transformers.TrainingArguments.use_ipex",description:`<strong>use_ipex</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Use Intel extension for PyTorch when it is available. <a href="https://github.com/intel/intel-extension-for-pytorch" rel="nofollow">IPEX | |
| installation</a>.`,name:"use_ipex"},{anchor:"transformers.TrainingArguments.bf16",description:`<strong>bf16</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to use bf16 16-bit (mixed) precision training instead of 32-bit training. Requires Ampere or higher | |
| NVIDIA architecture or using CPU (use_cpu) or Ascend NPU. This is an experimental API and it may change.`,name:"bf16"},{anchor:"transformers.TrainingArguments.fp16",description:`<strong>fp16</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to use fp16 16-bit (mixed) precision training instead of 32-bit training.`,name:"fp16"},{anchor:"transformers.TrainingArguments.fp16_opt_level",description:`<strong>fp16_opt_level</strong> (<code>str</code>, <em>optional</em>, defaults to ‘O1’) — | |
| For <code>fp16</code> training, Apex AMP optimization level selected in [‘O0’, ‘O1’, ‘O2’, and ‘O3’]. See details on | |
| the <a href="https://nvidia.github.io/apex/amp" rel="nofollow">Apex documentation</a>.`,name:"fp16_opt_level"},{anchor:"transformers.TrainingArguments.fp16_backend",description:`<strong>fp16_backend</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"auto"</code>) — | |
| This argument is deprecated. Use <code>half_precision_backend</code> instead.`,name:"fp16_backend"},{anchor:"transformers.TrainingArguments.half_precision_backend",description:`<strong>half_precision_backend</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"auto"</code>) — | |
| The backend to use for mixed precision training. Must be one of <code>"auto", "apex", "cpu_amp"</code>. <code>"auto"</code> will | |
| use CPU/CUDA AMP or APEX depending on the PyTorch version detected, while the other choices will force the | |
| requested backend.`,name:"half_precision_backend"},{anchor:"transformers.TrainingArguments.bf16_full_eval",description:`<strong>bf16_full_eval</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to use full bfloat16 evaluation instead of 32-bit. This will be faster and save memory but can harm | |
| metric values. This is an experimental API and it may change.`,name:"bf16_full_eval"},{anchor:"transformers.TrainingArguments.fp16_full_eval",description:`<strong>fp16_full_eval</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to use full float16 evaluation instead of 32-bit. This will be faster and save memory but can harm | |
| metric values.`,name:"fp16_full_eval"},{anchor:"transformers.TrainingArguments.tf32",description:`<strong>tf32</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether to enable the TF32 mode, available in Ampere and newer GPU architectures. The default value depends | |
| on PyTorch’s version default of <code>torch.backends.cuda.matmul.allow_tf32</code>. For more details please refer to | |
| the <a href="https://huggingface.co/docs/transformers/perf_train_gpu_one#tf32" rel="nofollow">TF32</a> documentation. This is an | |
| experimental API and it may change.`,name:"tf32"},{anchor:"transformers.TrainingArguments.local_rank",description:`<strong>local_rank</strong> (<code>int</code>, <em>optional</em>, defaults to -1) — | |
| Rank of the process during distributed training.`,name:"local_rank"},{anchor:"transformers.TrainingArguments.ddp_backend",description:`<strong>ddp_backend</strong> (<code>str</code>, <em>optional</em>) — | |
| The backend to use for distributed training. Must be one of <code>"nccl"</code>, <code>"mpi"</code>, <code>"ccl"</code>, <code>"gloo"</code>, <code>"hccl"</code>.`,name:"ddp_backend"},{anchor:"transformers.TrainingArguments.tpu_num_cores",description:`<strong>tpu_num_cores</strong> (<code>int</code>, <em>optional</em>) — | |
| When training on TPU, the number of TPU cores (automatically passed by launcher script).`,name:"tpu_num_cores"},{anchor:"transformers.TrainingArguments.dataloader_drop_last",description:`<strong>dataloader_drop_last</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to drop the last incomplete batch (if the length of the dataset is not divisible by the batch size) | |
| or not.`,name:"dataloader_drop_last"},{anchor:"transformers.TrainingArguments.eval_steps",description:`<strong>eval_steps</strong> (<code>int</code> or <code>float</code>, <em>optional</em>) — | |
| Number of update steps between two evaluations if <code>eval_strategy="steps"</code>. Will default to the same | |
| value as <code>logging_steps</code> if not set. Should be an integer or a float in range <code>[0,1)</code>. If smaller than 1, | |
| will be interpreted as ratio of total training steps.`,name:"eval_steps"},{anchor:"transformers.TrainingArguments.dataloader_num_workers",description:`<strong>dataloader_num_workers</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| Number of subprocesses to use for data loading (PyTorch only). 0 means that the data will be loaded in the | |
| main process.`,name:"dataloader_num_workers"},{anchor:"transformers.TrainingArguments.past_index",description:`<strong>past_index</strong> (<code>int</code>, <em>optional</em>, defaults to -1) — | |
| Some models like <a href="../model_doc/transformerxl">TransformerXL</a> or <a href="../model_doc/xlnet">XLNet</a> can make use of | |
| the past hidden states for their predictions. If this argument is set to a positive int, the <code>Trainer</code> will | |
| use the corresponding output (usually index 2) as the past state and feed it to the model at the next | |
| training step under the keyword argument <code>mems</code>.`,name:"past_index"},{anchor:"transformers.TrainingArguments.run_name",description:`<strong>run_name</strong> (<code>str</code>, <em>optional</em>, defaults to <code>output_dir</code>) — | |
| A descriptor for the run. Typically used for <a href="https://www.wandb.com/" rel="nofollow">wandb</a>, | |
| <a href="https://www.mlflow.org/" rel="nofollow">mlflow</a> and <a href="https://www.comet.com/site" rel="nofollow">comet</a> logging. If not specified, will | |
| be the same as <code>output_dir</code>.`,name:"run_name"},{anchor:"transformers.TrainingArguments.disable_tqdm",description:`<strong>disable_tqdm</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to disable the tqdm progress bars and table of metrics produced by | |
| <code>~notebook.NotebookTrainingTracker</code> in Jupyter Notebooks. Will default to <code>True</code> if the logging level is | |
| set to warn or lower (default), <code>False</code> otherwise.`,name:"disable_tqdm"},{anchor:"transformers.TrainingArguments.remove_unused_columns",description:`<strong>remove_unused_columns</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to automatically remove the columns unused by the model forward method.`,name:"remove_unused_columns"},{anchor:"transformers.TrainingArguments.label_names",description:`<strong>label_names</strong> (<code>List[str]</code>, <em>optional</em>) — | |
| The list of keys in your dictionary of inputs that correspond to the labels.</p> | |
| <p>Will eventually default to the list of argument names accepted by the model that contain the word “label”, | |
| except if the model used is one of the <code>XxxForQuestionAnswering</code> in which case it will also include the | |
| <code>["start_positions", "end_positions"]</code> keys.`,name:"label_names"},{anchor:"transformers.TrainingArguments.load_best_model_at_end",description:`<strong>load_best_model_at_end</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to load the best model found during training at the end of training. When this option is | |
| enabled, the best checkpoint will always be saved. See | |
| <a href="https://huggingface.co/docs/transformers/main_classes/trainer#transformers.TrainingArguments.save_total_limit" rel="nofollow"><code>save_total_limit</code></a> | |
| for more.</p> | |
| <div class="course-tip bg-gradient-to-br dark:bg-gradient-to-r before:border-green-500 dark:before:border-green-800 from-green-50 dark:from-gray-900 to-white dark:to-gray-950 border border-green-50 text-green-700 dark:text-gray-400"> | |
| <p>When set to <code>True</code>, the parameters <code>save_strategy</code> needs to be the same as <code>eval_strategy</code>, and in | |
| the case it is “steps”, <code>save_steps</code> must be a round multiple of <code>eval_steps</code>.</p> | |
| </div>`,name:"load_best_model_at_end"},{anchor:"transformers.TrainingArguments.metric_for_best_model",description:`<strong>metric_for_best_model</strong> (<code>str</code>, <em>optional</em>) — | |
| Use in conjunction with <code>load_best_model_at_end</code> to specify the metric to use to compare two different | |
| models. Must be the name of a metric returned by the evaluation with or without the prefix <code>"eval_"</code>. Will | |
| default to <code>"loss"</code> if unspecified and <code>load_best_model_at_end=True</code> (to use the evaluation loss).</p> | |
| <p>If you set this value, <code>greater_is_better</code> will default to <code>True</code>. Don’t forget to set it to <code>False</code> if | |
| your metric is better when lower.`,name:"metric_for_best_model"},{anchor:"transformers.TrainingArguments.greater_is_better",description:`<strong>greater_is_better</strong> (<code>bool</code>, <em>optional</em>) — | |
| Use in conjunction with <code>load_best_model_at_end</code> and <code>metric_for_best_model</code> to specify if better models | |
| should have a greater metric or not. Will default to:</p> | |
| <ul> | |
| <li><code>True</code> if <code>metric_for_best_model</code> is set to a value that doesn’t end in <code>"loss"</code>.</li> | |
| <li><code>False</code> if <code>metric_for_best_model</code> is not set, or set to a value that ends in <code>"loss"</code>.</li> | |
| </ul>`,name:"greater_is_better"},{anchor:"transformers.TrainingArguments.ignore_data_skip",description:`<strong>ignore_data_skip</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| When resuming training, whether or not to skip the epochs and batches to get the data loading at the same | |
| stage as in the previous training. If set to <code>True</code>, the training will begin faster (as that skipping step | |
| can take a long time) but will not yield the same results as the interrupted training would have.`,name:"ignore_data_skip"},{anchor:"transformers.TrainingArguments.fsdp",description:`<strong>fsdp</strong> (<code>bool</code>, <code>str</code> or list of <code>FSDPOption</code>, <em>optional</em>, defaults to <code>''</code>) — | |
| Use PyTorch Distributed Parallel Training (in distributed training only).</p> | |
| <p>A list of options along the following:</p> | |
| <ul> | |
| <li><code>"full_shard"</code>: Shard parameters, gradients and optimizer states.</li> | |
| <li><code>"shard_grad_op"</code>: Shard optimizer states and gradients.</li> | |
| <li><code>"hybrid_shard"</code>: Apply <code>FULL_SHARD</code> within a node, and replicate parameters across nodes.</li> | |
| <li><code>"hybrid_shard_zero2"</code>: Apply <code>SHARD_GRAD_OP</code> within a node, and replicate parameters across nodes.</li> | |
| <li><code>"offload"</code>: Offload parameters and gradients to CPUs (only compatible with <code>"full_shard"</code> and | |
| <code>"shard_grad_op"</code>).</li> | |
| <li><code>"auto_wrap"</code>: Automatically recursively wrap layers with FSDP using <code>default_auto_wrap_policy</code>.</li> | |
| </ul>`,name:"fsdp"},{anchor:"transformers.TrainingArguments.fsdp_config",description:`<strong>fsdp_config</strong> (<code>str</code> or <code>dict</code>, <em>optional</em>) — | |
| Config to be used with fsdp (Pytorch Distributed Parallel Training). The value is either a location of | |
| fsdp json config file (e.g., <code>fsdp_config.json</code>) or an already loaded json file as <code>dict</code>.</p> | |
| <p>A List of config and its options:</p> | |
| <ul> | |
| <li> | |
| <p>min_num_params (<code>int</code>, <em>optional</em>, defaults to <code>0</code>): | |
| FSDP’s minimum number of parameters for Default Auto Wrapping. (useful only when <code>fsdp</code> field is | |
| passed).</p> | |
| </li> | |
| <li> | |
| <p>transformer_layer_cls_to_wrap (<code>List[str]</code>, <em>optional</em>): | |
| List of transformer layer class names (case-sensitive) to wrap, e.g, <code>BertLayer</code>, <code>GPTJBlock</code>, | |
| <code>T5Block</code> … (useful only when <code>fsdp</code> flag is passed).</p> | |
| </li> | |
| <li> | |
| <p>backward_prefetch (<code>str</code>, <em>optional</em>) | |
| FSDP’s backward prefetch mode. Controls when to prefetch next set of parameters (useful only when | |
| <code>fsdp</code> field is passed).</p> | |
| <p>A list of options along the following:</p> | |
| <ul> | |
| <li><code>"backward_pre"</code> : Prefetches the next set of parameters before the current set of parameter’s | |
| gradient | |
| computation.</li> | |
| <li><code>"backward_post"</code> : This prefetches the next set of parameters after the current set of | |
| parameter’s | |
| gradient computation.</li> | |
| </ul> | |
| </li> | |
| <li> | |
| <p>forward_prefetch (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) | |
| FSDP’s forward prefetch mode (useful only when <code>fsdp</code> field is passed). | |
| If <code>"True"</code>, then FSDP explicitly prefetches the next upcoming all-gather while executing in the | |
| forward pass.</p> | |
| </li> | |
| <li> | |
| <p>limit_all_gathers (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) | |
| FSDP’s limit_all_gathers (useful only when <code>fsdp</code> field is passed). | |
| If <code>"True"</code>, FSDP explicitly synchronizes the CPU thread to prevent too many in-flight | |
| all-gathers.</p> | |
| </li> | |
| <li> | |
| <p>use_orig_params (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) | |
| If <code>"True"</code>, allows non-uniform <code>requires_grad</code> during init, which means support for interspersed | |
| frozen and trainable paramteres. Useful in cases such as parameter-efficient fine-tuning. Please | |
| refer this | |
| [blog](<a href="https://dev-discuss.pytorch.org/t/rethinking-pytorch-fully-sharded-data-parallel-fsdp-from-first-principles/1019" rel="nofollow">https://dev-discuss.pytorch.org/t/rethinking-pytorch-fully-sharded-data-parallel-fsdp-from-first-principles/1019</a></p> | |
| </li> | |
| <li> | |
| <p>sync_module_states (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) | |
| If <code>"True"</code>, each individually wrapped FSDP unit will broadcast module parameters from rank 0 to | |
| ensure they are the same across all ranks after initialization</p> | |
| </li> | |
| <li> | |
| <p>cpu_ram_efficient_loading (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) | |
| If <code>"True"</code>, only the first process loads the pretrained model checkpoint while all other processes | |
| have empty weights. When this setting as <code>"True"</code>, <code>sync_module_states</code> also must to be <code>"True"</code>, | |
| otherwise all the processes except the main process would have random weights leading to unexpected | |
| behaviour during training.</p> | |
| </li> | |
| <li> | |
| <p>activation_checkpointing (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>): | |
| If <code>"True"</code>, activation checkpointing is a technique to reduce memory usage by clearing activations of | |
| certain layers and recomputing them during a backward pass. Effectively, this trades extra | |
| computation time for reduced memory usage.</p> | |
| </li> | |
| <li> | |
| <p>xla (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>): | |
| Whether to use PyTorch/XLA Fully Sharded Data Parallel Training. This is an experimental feature | |
| and its API may evolve in the future.</p> | |
| </li> | |
| <li> | |
| <p>xla_fsdp_settings (<code>dict</code>, <em>optional</em>) | |
| The value is a dictionary which stores the XLA FSDP wrapping parameters.</p> | |
| <p>For a complete list of options, please see <a href="https://github.com/pytorch/xla/blob/master/torch_xla/distributed/fsdp/xla_fully_sharded_data_parallel.py" rel="nofollow">here</a>.</p> | |
| </li> | |
| <li> | |
| <p>xla_fsdp_grad_ckpt (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>): | |
| Will use gradient checkpointing over each nested XLA FSDP wrapped layer. This setting can only be | |
| used when the xla flag is set to true, and an auto wrapping policy is specified through | |
| fsdp_min_num_params or fsdp_transformer_layer_cls_to_wrap.</p> | |
| </li> | |
| </ul>`,name:"fsdp_config"},{anchor:"transformers.TrainingArguments.deepspeed",description:`<strong>deepspeed</strong> (<code>str</code> or <code>dict</code>, <em>optional</em>) — | |
| Use <a href="https://github.com/microsoft/deepspeed" rel="nofollow">Deepspeed</a>. This is an experimental feature and its API may | |
| evolve in the future. The value is either the location of DeepSpeed json config file (e.g., | |
| <code>ds_config.json</code>) or an already loaded json file as a <code>dict</code>”</p> | |
| <div class="course-tip course-tip-orange bg-gradient-to-br dark:bg-gradient-to-r before:border-orange-500 dark:before:border-orange-800 from-orange-50 dark:from-gray-900 to-white dark:to-gray-950 border border-orange-50 text-orange-700 dark:text-gray-400"> | |
| If enabling any Zero-init, make sure that your model is not initialized until | |
| *after* initializing the \`TrainingArguments\`, else it will not be applied. | |
| </div>`,name:"deepspeed"},{anchor:"transformers.TrainingArguments.accelerator_config",description:`<strong>accelerator_config</strong> (<code>str</code>, <code>dict</code>, or <code>AcceleratorConfig</code>, <em>optional</em>) — | |
| Config to be used with the internal <code>Accelerator</code> implementation. The value is either a location of | |
| accelerator json config file (e.g., <code>accelerator_config.json</code>), an already loaded json file as <code>dict</code>, | |
| or an instance of <code>AcceleratorConfig</code>.</p> | |
| <p>A list of config and its options:</p> | |
| <ul> | |
| <li>split_batches (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>): | |
| Whether or not the accelerator should split the batches yielded by the dataloaders across the devices. If | |
| <code>True</code> the actual batch size used will be the same on any kind of distributed processes, but it must be a | |
| round multiple of the <code>num_processes</code> you are using. If <code>False</code>, actual batch size used will be the one set | |
| in your script multiplied by the number of processes.</li> | |
| <li>dispatch_batches (<code>bool</code>, <em>optional</em>): | |
| If set to <code>True</code>, the dataloader prepared by the Accelerator is only iterated through on the main process | |
| and then the batches are split and broadcast to each process. Will default to <code>True</code> for <code>DataLoader</code> whose | |
| underlying dataset is an <code>IterableDataset</code>, <code>False</code> otherwise.</li> | |
| <li>even_batches (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>): | |
| If set to <code>True</code>, in cases where the total batch size across all processes does not exactly divide the | |
| dataset, samples at the start of the dataset will be duplicated so the batch can be divided equally among | |
| all workers.</li> | |
| <li>use_seedable_sampler (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>): | |
| Whether or not use a fully seedable random sampler (<code>accelerate.data_loader.SeedableRandomSampler</code>). Ensures | |
| training results are fully reproducable using a different sampling technique. While seed-to-seed results | |
| may differ, on average the differences are neglible when using multiple different seeds to compare. Should | |
| also be ran with <code>~utils.set_seed</code> for the best results.</li> | |
| <li>use_configured_state (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>): | |
| Whether or not to use a pre-configured <code>AcceleratorState</code> or <code>PartialState</code> defined before calling <code>TrainingArguments</code>. | |
| If <code>True</code>, an <code>Accelerator</code> or <code>PartialState</code> must be initialized. Note that by doing so, this could lead to issues | |
| with hyperparameter tuning.</li> | |
| </ul>`,name:"accelerator_config"},{anchor:"transformers.TrainingArguments.label_smoothing_factor",description:`<strong>label_smoothing_factor</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| The label smoothing factor to use. Zero means no label smoothing, otherwise the underlying onehot-encoded | |
| labels are changed from 0s and 1s to <code>label_smoothing_factor/num_labels</code> and <code>1 - label_smoothing_factor + label_smoothing_factor/num_labels</code> respectively.`,name:"label_smoothing_factor"},{anchor:"transformers.TrainingArguments.debug",description:`<strong>debug</strong> (<code>str</code> or list of <code>DebugOption</code>, <em>optional</em>, defaults to <code>""</code>) — | |
| Enable one or more debug features. This is an experimental feature.</p> | |
| <p>Possible options are:</p> | |
| <ul> | |
| <li><code>"underflow_overflow"</code>: detects overflow in model’s input/outputs and reports the last frames that led to | |
| the event</li> | |
| <li><code>"tpu_metrics_debug"</code>: print debug metrics on TPU</li> | |
| </ul> | |
| <p>The options should be separated by whitespaces.`,name:"debug"},{anchor:"transformers.TrainingArguments.optim",description:`<strong>optim</strong> (<code>str</code> or <code>training_args.OptimizerNames</code>, <em>optional</em>, defaults to <code>"adamw_torch"</code>) — | |
| The optimizer to use, such as “adamw_hf”, “adamw_torch”, “adamw_torch_fused”, “adamw_apex_fused”, “adamw_anyprecision”, | |
| “adafactor”. See <code>OptimizerNames</code> in <a href="https://github.com/huggingface/transformers/blob/main/src/transformers/training_args.py" rel="nofollow">training_args.py</a> | |
| for a full list of optimizers.`,name:"optim"},{anchor:"transformers.TrainingArguments.optim_args",description:`<strong>optim_args</strong> (<code>str</code>, <em>optional</em>) — | |
| Optional arguments that are supplied to optimizers such as AnyPrecisionAdamW, AdEMAMix, and GaLore.`,name:"optim_args"},{anchor:"transformers.TrainingArguments.group_by_length",description:`<strong>group_by_length</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to group together samples of roughly the same length in the training dataset (to minimize | |
| padding applied and be more efficient). Only useful if applying dynamic padding.`,name:"group_by_length"},{anchor:"transformers.TrainingArguments.length_column_name",description:`<strong>length_column_name</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"length"</code>) — | |
| Column name for precomputed lengths. If the column exists, grouping by length will use these values rather | |
| than computing them on train startup. Ignored unless <code>group_by_length</code> is <code>True</code> and the dataset is an | |
| instance of <code>Dataset</code>.`,name:"length_column_name"},{anchor:"transformers.TrainingArguments.report_to",description:`<strong>report_to</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>, defaults to <code>"all"</code>) — | |
| The list of integrations to report the results and logs to. Supported platforms are <code>"azure_ml"</code>, | |
| <code>"clearml"</code>, <code>"codecarbon"</code>, <code>"comet_ml"</code>, <code>"dagshub"</code>, <code>"dvclive"</code>, <code>"flyte"</code>, <code>"mlflow"</code>, <code>"neptune"</code>, | |
| <code>"tensorboard"</code>, and <code>"wandb"</code>. Use <code>"all"</code> to report to all integrations installed, <code>"none"</code> for no | |
| integrations.`,name:"report_to"},{anchor:"transformers.TrainingArguments.ddp_find_unused_parameters",description:`<strong>ddp_find_unused_parameters</strong> (<code>bool</code>, <em>optional</em>) — | |
| When using distributed training, the value of the flag <code>find_unused_parameters</code> passed to | |
| <code>DistributedDataParallel</code>. Will default to <code>False</code> if gradient checkpointing is used, <code>True</code> otherwise.`,name:"ddp_find_unused_parameters"},{anchor:"transformers.TrainingArguments.ddp_bucket_cap_mb",description:`<strong>ddp_bucket_cap_mb</strong> (<code>int</code>, <em>optional</em>) — | |
| When using distributed training, the value of the flag <code>bucket_cap_mb</code> passed to <code>DistributedDataParallel</code>.`,name:"ddp_bucket_cap_mb"},{anchor:"transformers.TrainingArguments.ddp_broadcast_buffers",description:`<strong>ddp_broadcast_buffers</strong> (<code>bool</code>, <em>optional</em>) — | |
| When using distributed training, the value of the flag <code>broadcast_buffers</code> passed to | |
| <code>DistributedDataParallel</code>. Will default to <code>False</code> if gradient checkpointing is used, <code>True</code> otherwise.`,name:"ddp_broadcast_buffers"},{anchor:"transformers.TrainingArguments.dataloader_pin_memory",description:`<strong>dataloader_pin_memory</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether you want to pin memory in data loaders or not. Will default to <code>True</code>.`,name:"dataloader_pin_memory"},{anchor:"transformers.TrainingArguments.dataloader_persistent_workers",description:`<strong>dataloader_persistent_workers</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| If True, the data loader will not shut down the worker processes after a dataset has been consumed once. | |
| This allows to maintain the workers Dataset instances alive. Can potentially speed up training, but will | |
| increase RAM usage. Will default to <code>False</code>.`,name:"dataloader_persistent_workers"},{anchor:"transformers.TrainingArguments.dataloader_prefetch_factor",description:`<strong>dataloader_prefetch_factor</strong> (<code>int</code>, <em>optional</em>) — | |
| Number of batches loaded in advance by each worker. | |
| 2 means there will be a total of 2 * num_workers batches prefetched across all workers.`,name:"dataloader_prefetch_factor"},{anchor:"transformers.TrainingArguments.skip_memory_metrics",description:`<strong>skip_memory_metrics</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to skip adding of memory profiler reports to metrics. This is skipped by default because it slows | |
| down the training and evaluation speed.`,name:"skip_memory_metrics"},{anchor:"transformers.TrainingArguments.push_to_hub",description:`<strong>push_to_hub</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to push the model to the Hub every time the model is saved. If this is activated, | |
| <code>output_dir</code> will begin a git directory synced with the repo (determined by <code>hub_model_id</code>) and the content | |
| will be pushed each time a save is triggered (depending on your <code>save_strategy</code>). Calling | |
| <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer.save_model">save_model()</a> will also trigger a push.</p> | |
| <div class="course-tip course-tip-orange bg-gradient-to-br dark:bg-gradient-to-r before:border-orange-500 dark:before:border-orange-800 from-orange-50 dark:from-gray-900 to-white dark:to-gray-950 border border-orange-50 text-orange-700 dark:text-gray-400"> | |
| <p>If <code>output_dir</code> exists, it needs to be a local clone of the repository to which the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a> will be | |
| pushed.</p> | |
| </div>`,name:"push_to_hub"},{anchor:"transformers.TrainingArguments.resume_from_checkpoint",description:`<strong>resume_from_checkpoint</strong> (<code>str</code>, <em>optional</em>) — | |
| The path to a folder with a valid checkpoint for your model. This argument is not directly used by | |
| <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>, it’s intended to be used by your training/evaluation scripts instead. See the <a href="https://github.com/huggingface/transformers/tree/main/examples" rel="nofollow">example | |
| scripts</a> for more details.`,name:"resume_from_checkpoint"},{anchor:"transformers.TrainingArguments.hub_model_id",description:`<strong>hub_model_id</strong> (<code>str</code>, <em>optional</em>) — | |
| The name of the repository to keep in sync with the local <em>output_dir</em>. It can be a simple model ID in | |
| which case the model will be pushed in your namespace. Otherwise it should be the whole repository name, | |
| for instance <code>"user_name/model"</code>, which allows you to push to an organization you are a member of with | |
| <code>"organization_name/model"</code>. Will default to <code>user_name/output_dir_name</code> with <em>output_dir_name</em> being the | |
| name of <code>output_dir</code>.</p> | |
| <p>Will default to the name of <code>output_dir</code>.`,name:"hub_model_id"},{anchor:"transformers.TrainingArguments.hub_strategy",description:`<strong>hub_strategy</strong> (<code>str</code> or <code>HubStrategy</code>, <em>optional</em>, defaults to <code>"every_save"</code>) — | |
| Defines the scope of what is pushed to the Hub and when. Possible values are:</p> | |
| <ul> | |
| <li><code>"end"</code>: push the model, its configuration, the processing class e.g. tokenizer (if passed along to the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>) and a | |
| draft of a model card when the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer.save_model">save_model()</a> method is called.</li> | |
| <li><code>"every_save"</code>: push the model, its configuration, the processing class e.g. tokenizer (if passed along to the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>) and | |
| a draft of a model card each time there is a model save. The pushes are asynchronous to not block | |
| training, and in case the save are very frequent, a new push is only attempted if the previous one is | |
| finished. A last push is made with the final model at the end of training.</li> | |
| <li><code>"checkpoint"</code>: like <code>"every_save"</code> but the latest checkpoint is also pushed in a subfolder named | |
| last-checkpoint, allowing you to resume training easily with | |
| <code>trainer.train(resume_from_checkpoint="last-checkpoint")</code>.</li> | |
| <li><code>"all_checkpoints"</code>: like <code>"checkpoint"</code> but all checkpoints are pushed like they appear in the output | |
| folder (so you will get one checkpoint folder per folder in your final repository)</li> | |
| </ul>`,name:"hub_strategy"},{anchor:"transformers.TrainingArguments.hub_token",description:`<strong>hub_token</strong> (<code>str</code>, <em>optional</em>) — | |
| The token to use to push the model to the Hub. Will default to the token in the cache folder obtained with | |
| <code>huggingface-cli login</code>.`,name:"hub_token"},{anchor:"transformers.TrainingArguments.hub_private_repo",description:`<strong>hub_private_repo</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| If True, the Hub repo will be set to private.`,name:"hub_private_repo"},{anchor:"transformers.TrainingArguments.hub_always_push",description:`<strong>hub_always_push</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Unless this is <code>True</code>, the <code>Trainer</code> will skip pushing a checkpoint when the previous push is not finished.`,name:"hub_always_push"},{anchor:"transformers.TrainingArguments.gradient_checkpointing",description:`<strong>gradient_checkpointing</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| If True, use gradient checkpointing to save memory at the expense of slower backward pass.`,name:"gradient_checkpointing"},{anchor:"transformers.TrainingArguments.gradient_checkpointing_kwargs",description:`<strong>gradient_checkpointing_kwargs</strong> (<code>dict</code>, <em>optional</em>, defaults to <code>None</code>) — | |
| Key word arguments to be passed to the <code>gradient_checkpointing_enable</code> method.`,name:"gradient_checkpointing_kwargs"},{anchor:"transformers.TrainingArguments.include_inputs_for_metrics",description:`<strong>include_inputs_for_metrics</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| This argument is deprecated. Use <code>include_for_metrics</code> instead, e.g, <code>include_for_metrics = ["inputs"]</code>.`,name:"include_inputs_for_metrics"},{anchor:"transformers.TrainingArguments.include_for_metrics",description:`<strong>include_for_metrics</strong> (<code>List[str]</code>, <em>optional</em>, defaults to <code>[]</code>) — | |
| Include additional data in the <code>compute_metrics</code> function if needed for metrics computation. | |
| Possible options to add to <code>include_for_metrics</code> list:</p> | |
| <ul> | |
| <li><code>"inputs"</code>: Input data passed to the model, intended for calculating input dependent metrics.</li> | |
| <li><code>"loss"</code>: Loss values computed during evaluation, intended for calculating loss dependent metrics.</li> | |
| </ul>`,name:"include_for_metrics"},{anchor:"transformers.TrainingArguments.eval_do_concat_batches",description:`<strong>eval_do_concat_batches</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to recursively concat inputs/losses/labels/predictions across batches. If <code>False</code>, | |
| will instead store them as lists, with each batch kept separate.`,name:"eval_do_concat_batches"},{anchor:"transformers.TrainingArguments.auto_find_batch_size",description:`<strong>auto_find_batch_size</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to find a batch size that will fit into memory automatically through exponential decay, avoiding | |
| CUDA Out-of-Memory errors. Requires accelerate to be installed (<code>pip install accelerate</code>)`,name:"auto_find_batch_size"},{anchor:"transformers.TrainingArguments.full_determinism",description:`<strong>full_determinism</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| If <code>True</code>, <code>enable_full_determinism()</code> is called instead of <code>set_seed()</code> to ensure reproducible results in | |
| distributed training. Important: this will negatively impact the performance, so only use it for debugging.`,name:"full_determinism"},{anchor:"transformers.TrainingArguments.torchdynamo",description:`<strong>torchdynamo</strong> (<code>str</code>, <em>optional</em>) — | |
| If set, the backend compiler for TorchDynamo. Possible choices are <code>"eager"</code>, <code>"aot_eager"</code>, <code>"inductor"</code>, | |
| <code>"nvfuser"</code>, <code>"aot_nvfuser"</code>, <code>"aot_cudagraphs"</code>, <code>"ofi"</code>, <code>"fx2trt"</code>, <code>"onnxrt"</code> and <code>"ipex"</code>.`,name:"torchdynamo"},{anchor:"transformers.TrainingArguments.ray_scope",description:`<strong>ray_scope</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"last"</code>) — | |
| The scope to use when doing hyperparameter search with Ray. By default, <code>"last"</code> will be used. Ray will | |
| then use the last checkpoint of all trials, compare those, and select the best one. However, other options | |
| are also available. See the <a href="https://docs.ray.io/en/latest/tune/api_docs/analysis.html#ray.tune.ExperimentAnalysis.get_best_trial" rel="nofollow">Ray documentation</a> for | |
| more options.`,name:"ray_scope"},{anchor:"transformers.TrainingArguments.ddp_timeout",description:`<strong>ddp_timeout</strong> (<code>int</code>, <em>optional</em>, defaults to 1800) — | |
| The timeout for <code>torch.distributed.init_process_group</code> calls, used to avoid GPU socket timeouts when | |
| performing slow operations in distributed runnings. Please refer the [PyTorch documentation] | |
| (<a href="https://pytorch.org/docs/stable/distributed.html#torch.distributed.init_process_group" rel="nofollow">https://pytorch.org/docs/stable/distributed.html#torch.distributed.init_process_group</a>) for more | |
| information.`,name:"ddp_timeout"},{anchor:"transformers.TrainingArguments.use_mps_device",description:`<strong>use_mps_device</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| This argument is deprecated.<code>mps</code> device will be used if it is available similar to <code>cuda</code> device.`,name:"use_mps_device"},{anchor:"transformers.TrainingArguments.torch_compile",description:`<strong>torch_compile</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to compile the model using PyTorch 2.0 | |
| <a href="https://pytorch.org/get-started/pytorch-2.0/" rel="nofollow"><code>torch.compile</code></a>.</p> | |
| <p>This will use the best defaults for the <a href="https://pytorch.org/docs/stable/generated/torch.compile.html?highlight=torch+compile#torch.compile" rel="nofollow"><code>torch.compile</code> | |
| API</a>. | |
| You can customize the defaults with the argument <code>torch_compile_backend</code> and <code>torch_compile_mode</code> but we | |
| don’t guarantee any of them will work as the support is progressively rolled in in PyTorch.</p> | |
| <p>This flag and the whole compile API is experimental and subject to change in future releases.`,name:"torch_compile"},{anchor:"transformers.TrainingArguments.torch_compile_backend",description:`<strong>torch_compile_backend</strong> (<code>str</code>, <em>optional</em>) — | |
| The backend to use in <code>torch.compile</code>. If set to any value, <code>torch_compile</code> will be set to <code>True</code>.</p> | |
| <p>Refer to the PyTorch doc for possible values and note that they may change across PyTorch versions.</p> | |
| <p>This flag is experimental and subject to change in future releases.`,name:"torch_compile_backend"},{anchor:"transformers.TrainingArguments.torch_compile_mode",description:`<strong>torch_compile_mode</strong> (<code>str</code>, <em>optional</em>) — | |
| The mode to use in <code>torch.compile</code>. If set to any value, <code>torch_compile</code> will be set to <code>True</code>.</p> | |
| <p>Refer to the PyTorch doc for possible values and note that they may change across PyTorch versions.</p> | |
| <p>This flag is experimental and subject to change in future releases.`,name:"torch_compile_mode"},{anchor:"transformers.TrainingArguments.split_batches",description:`<strong>split_batches</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not the accelerator should split the batches yielded by the dataloaders across the devices | |
| during distributed training. If</p> | |
| <p>set to <code>True</code>, the actual batch size used will be the same on any kind of distributed processes, but it | |
| must be a</p> | |
| <p>round multiple of the number of processes you are using (such as GPUs).`,name:"split_batches"},{anchor:"transformers.TrainingArguments.include_tokens_per_second",description:`<strong>include_tokens_per_second</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to compute the number of tokens per second per device for training speed metrics.</p> | |
| <p>This will iterate over the entire training dataloader once beforehand,</p> | |
| <p>and will slow down the entire process.`,name:"include_tokens_per_second"},{anchor:"transformers.TrainingArguments.include_num_input_tokens_seen",description:`<strong>include_num_input_tokens_seen</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to track the number of input tokens seen throughout training.</p> | |
| <p>May be slower in distributed training as gather operations must be called.`,name:"include_num_input_tokens_seen"},{anchor:"transformers.TrainingArguments.neftune_noise_alpha",description:`<strong>neftune_noise_alpha</strong> (<code>Optional[float]</code>) — | |
| If not <code>None</code>, this will activate NEFTune noise embeddings. This can drastically improve model performance | |
| for instruction fine-tuning. Check out the <a href="https://arxiv.org/abs/2310.05914" rel="nofollow">original paper</a> and the | |
| <a href="https://github.com/neelsjain/NEFTune" rel="nofollow">original code</a>. Support transformers <code>PreTrainedModel</code> and also | |
| <code>PeftModel</code> from peft. The original paper used values in the range [5.0, 15.0].`,name:"neftune_noise_alpha"},{anchor:"transformers.TrainingArguments.optim_target_modules",description:`<strong>optim_target_modules</strong> (<code>Union[str, List[str]]</code>, <em>optional</em>) — | |
| The target modules to optimize, i.e. the module names that you would like to train, right now this is used only for GaLore algorithm | |
| <a href="https://arxiv.org/abs/2403.03507" rel="nofollow">https://arxiv.org/abs/2403.03507</a> | |
| See: <a href="https://github.com/jiaweizzhao/GaLore" rel="nofollow">https://github.com/jiaweizzhao/GaLore</a> for more details. You need to make sure to pass a valid GaloRe | |
| optimizer, e.g. one of: “galore_adamw”, “galore_adamw_8bit”, “galore_adafactor” and make sure that the target modules are <code>nn.Linear</code> modules | |
| only.`,name:"optim_target_modules"},{anchor:"transformers.TrainingArguments.batch_eval_metrics",description:`<strong>batch_eval_metrics</strong> (<code>Optional[bool]</code>, defaults to <code>False</code>) — | |
| If set to <code>True</code>, evaluation will call compute_metrics at the end of each batch to accumulate statistics | |
| rather than saving all eval logits in memory. When set to <code>True</code>, you must pass a compute_metrics function | |
| that takes a boolean argument <code>compute_result</code>, which when passed <code>True</code>, will trigger the final global | |
| summary statistics from the batch-level summary statistics you’ve accumulated over the evaluation set.`,name:"batch_eval_metrics"},{anchor:"transformers.TrainingArguments.eval_on_start",description:`<strong>eval_on_start</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to perform a evaluation step (sanity check) before the training to ensure the validation steps works correctly.`,name:"eval_on_start"},{anchor:"transformers.TrainingArguments.eval_use_gather_object",description:`<strong>eval_use_gather_object</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to run recursively gather object in a nested list/tuple/dictionary of objects from all devices. This should only be enabled if users are not just returning tensors, and this is actively discouraged by PyTorch.`,name:"eval_use_gather_object"},{anchor:"transformers.TrainingArguments.use_liger_kernel",description:`<strong>use_liger_kernel</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether enable <a href="https://github.com/linkedin/Liger-Kernel" rel="nofollow">Liger</a> Kernel for LLM model training. | |
| It can effectively increase multi-GPU training throughput by ~20% and reduces memory usage by ~60%, works out of the box with | |
| flash attention, PyTorch FSDP, and Microsoft DeepSpeed. Currently, it supports llama, mistral, mixtral and gemma models.`,name:"use_liger_kernel"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L221"}}),_o=new x({props:{name:"get_process_log_level",anchor:"transformers.TrainingArguments.get_process_log_level",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2377"}}),bo=new x({props:{name:"get_warmup_steps",anchor:"transformers.TrainingArguments.get_warmup_steps",parameters:[{name:"num_training_steps",val:": int"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2466"}}),vo=new x({props:{name:"main_process_first",anchor:"transformers.TrainingArguments.main_process_first",parameters:[{name:"local",val:" = True"},{name:"desc",val:" = 'work'"}],parametersDescription:[{anchor:"transformers.TrainingArguments.main_process_first.local",description:`<strong>local</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| if <code>True</code> first means process of rank 0 of each node if <code>False</code> first means process of rank 0 of node | |
| rank 0 In multi-node environment with a shared filesystem you most likely will want to use | |
| <code>local=False</code> so that only the main process of the first node will do the processing. If however, the | |
| filesystem is not shared, then the main process of each node will need to do the processing, which is | |
| the default behavior.`,name:"local"},{anchor:"transformers.TrainingArguments.main_process_first.desc",description:`<strong>desc</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"work"</code>) — | |
| a work description to be used in debug logs`,name:"desc"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2415"}}),yo=new x({props:{name:"set_dataloader",anchor:"transformers.TrainingArguments.set_dataloader",parameters:[{name:"train_batch_size",val:": int = 8"},{name:"eval_batch_size",val:": int = 8"},{name:"drop_last",val:": bool = False"},{name:"num_workers",val:": int = 0"},{name:"pin_memory",val:": bool = True"},{name:"persistent_workers",val:": bool = False"},{name:"prefetch_factor",val:": Optional = None"},{name:"auto_find_batch_size",val:": bool = False"},{name:"ignore_data_skip",val:": bool = False"},{name:"sampler_seed",val:": Optional = None"}],parametersDescription:[{anchor:"transformers.TrainingArguments.set_dataloader.drop_last",description:`<strong>drop_last</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to drop the last incomplete batch (if the length of the dataset is not divisible by the batch | |
| size) or not.`,name:"drop_last"},{anchor:"transformers.TrainingArguments.set_dataloader.num_workers",description:`<strong>num_workers</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| Number of subprocesses to use for data loading (PyTorch only). 0 means that the data will be loaded in | |
| the main process.`,name:"num_workers"},{anchor:"transformers.TrainingArguments.set_dataloader.pin_memory",description:`<strong>pin_memory</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether you want to pin memory in data loaders or not. Will default to <code>True</code>.`,name:"pin_memory"},{anchor:"transformers.TrainingArguments.set_dataloader.persistent_workers",description:`<strong>persistent_workers</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| If True, the data loader will not shut down the worker processes after a dataset has been consumed | |
| once. This allows to maintain the workers Dataset instances alive. Can potentially speed up training, | |
| but will increase RAM usage. Will default to <code>False</code>.`,name:"persistent_workers"},{anchor:"transformers.TrainingArguments.set_dataloader.prefetch_factor",description:`<strong>prefetch_factor</strong> (<code>int</code>, <em>optional</em>) — | |
| Number of batches loaded in advance by each worker. | |
| 2 means there will be a total of 2 * num_workers batches prefetched across all workers.`,name:"prefetch_factor"},{anchor:"transformers.TrainingArguments.set_dataloader.auto_find_batch_size",description:`<strong>auto_find_batch_size</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to find a batch size that will fit into memory automatically through exponential decay, | |
| avoiding CUDA Out-of-Memory errors. Requires accelerate to be installed (<code>pip install accelerate</code>)`,name:"auto_find_batch_size"},{anchor:"transformers.TrainingArguments.set_dataloader.ignore_data_skip",description:`<strong>ignore_data_skip</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| When resuming training, whether or not to skip the epochs and batches to get the data loading at the | |
| same stage as in the previous training. If set to <code>True</code>, the training will begin faster (as that | |
| skipping step can take a long time) but will not yield the same results as the interrupted training | |
| would have.`,name:"ignore_data_skip"},{anchor:"transformers.TrainingArguments.set_dataloader.sampler_seed",description:`<strong>sampler_seed</strong> (<code>int</code>, <em>optional</em>) — | |
| Random seed to be used with data samplers. If not set, random generators for data sampling will use the | |
| same seed as <code>self.seed</code>. This can be used to ensure reproducibility of data sampling, independent of | |
| the model seed.`,name:"sampler_seed"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2995"}}),Qe=new be({props:{anchor:"transformers.TrainingArguments.set_dataloader.example",$$slots:{default:[$m]},$$scope:{ctx:C}}}),To=new x({props:{name:"set_evaluate",anchor:"transformers.TrainingArguments.set_evaluate",parameters:[{name:"strategy",val:": Union = 'no'"},{name:"steps",val:": int = 500"},{name:"batch_size",val:": int = 8"},{name:"accumulation_steps",val:": Optional = None"},{name:"delay",val:": Optional = None"},{name:"loss_only",val:": bool = False"},{name:"jit_mode",val:": bool = False"}],parametersDescription:[{anchor:"transformers.TrainingArguments.set_evaluate.strategy",description:`<strong>strategy</strong> (<code>str</code> or <code>IntervalStrategy</code>, <em>optional</em>, defaults to <code>"no"</code>) — | |
| The evaluation strategy to adopt during training. Possible values are:</p> | |
| <ul> | |
| <li><code>"no"</code>: No evaluation is done during training.</li> | |
| <li><code>"steps"</code>: Evaluation is done (and logged) every <code>steps</code>.</li> | |
| <li><code>"epoch"</code>: Evaluation is done at the end of each epoch.</li> | |
| </ul> | |
| <p>Setting a <code>strategy</code> different from <code>"no"</code> will set <code>self.do_eval</code> to <code>True</code>.`,name:"strategy"},{anchor:"transformers.TrainingArguments.set_evaluate.steps",description:`<strong>steps</strong> (<code>int</code>, <em>optional</em>, defaults to 500) — | |
| Number of update steps between two evaluations if <code>strategy="steps"</code>.`,name:"steps"},{anchor:"transformers.TrainingArguments.set_evaluate.batch_size",description:`<strong>batch_size</strong> (<code>int</code> <em>optional</em>, defaults to 8) — | |
| The batch size per device (GPU/TPU core/CPU…) used for evaluation.`,name:"batch_size"},{anchor:"transformers.TrainingArguments.set_evaluate.accumulation_steps",description:`<strong>accumulation_steps</strong> (<code>int</code>, <em>optional</em>) — | |
| Number of predictions steps to accumulate the output tensors for, before moving the results to the CPU. | |
| If left unset, the whole predictions are accumulated on GPU/TPU before being moved to the CPU (faster | |
| but requires more memory).`,name:"accumulation_steps"},{anchor:"transformers.TrainingArguments.set_evaluate.delay",description:`<strong>delay</strong> (<code>float</code>, <em>optional</em>) — | |
| Number of epochs or steps to wait for before the first evaluation can be performed, depending on the | |
| eval_strategy.`,name:"delay"},{anchor:"transformers.TrainingArguments.set_evaluate.loss_only",description:`<strong>loss_only</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Ignores all outputs except the loss.`,name:"loss_only"},{anchor:"transformers.TrainingArguments.set_evaluate.jit_mode",description:`<strong>jit_mode</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to use PyTorch jit trace for inference.`,name:"jit_mode"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2604"}}),Ke=new be({props:{anchor:"transformers.TrainingArguments.set_evaluate.example",$$slots:{default:[Am]},$$scope:{ctx:C}}}),wo=new x({props:{name:"set_logging",anchor:"transformers.TrainingArguments.set_logging",parameters:[{name:"strategy",val:": Union = 'steps'"},{name:"steps",val:": int = 500"},{name:"report_to",val:": Union = 'none'"},{name:"level",val:": str = 'passive'"},{name:"first_step",val:": bool = False"},{name:"nan_inf_filter",val:": bool = False"},{name:"on_each_node",val:": bool = False"},{name:"replica_level",val:": str = 'passive'"}],parametersDescription:[{anchor:"transformers.TrainingArguments.set_logging.strategy",description:`<strong>strategy</strong> (<code>str</code> or <code>IntervalStrategy</code>, <em>optional</em>, defaults to <code>"steps"</code>) — | |
| The logging strategy to adopt during training. Possible values are:</p> | |
| <ul> | |
| <li><code>"no"</code>: No logging is done during training.</li> | |
| <li><code>"epoch"</code>: Logging is done at the end of each epoch.</li> | |
| <li><code>"steps"</code>: Logging is done every <code>logging_steps</code>.</li> | |
| </ul>`,name:"strategy"},{anchor:"transformers.TrainingArguments.set_logging.steps",description:`<strong>steps</strong> (<code>int</code>, <em>optional</em>, defaults to 500) — | |
| Number of update steps between two logs if <code>strategy="steps"</code>.`,name:"steps"},{anchor:"transformers.TrainingArguments.set_logging.level",description:`<strong>level</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"passive"</code>) — | |
| Logger log level to use on the main process. Possible choices are the log levels as strings: <code>"debug"</code>, | |
| <code>"info"</code>, <code>"warning"</code>, <code>"error"</code> and <code>"critical"</code>, plus a <code>"passive"</code> level which doesn’t set anything | |
| and lets the application set the level.`,name:"level"},{anchor:"transformers.TrainingArguments.set_logging.report_to",description:`<strong>report_to</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>, defaults to <code>"all"</code>) — | |
| The list of integrations to report the results and logs to. Supported platforms are <code>"azure_ml"</code>, | |
| <code>"clearml"</code>, <code>"codecarbon"</code>, <code>"comet_ml"</code>, <code>"dagshub"</code>, <code>"dvclive"</code>, <code>"flyte"</code>, <code>"mlflow"</code>, | |
| <code>"neptune"</code>, <code>"tensorboard"</code>, and <code>"wandb"</code>. Use <code>"all"</code> to report to all integrations installed, | |
| <code>"none"</code> for no integrations.`,name:"report_to"},{anchor:"transformers.TrainingArguments.set_logging.first_step",description:`<strong>first_step</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to log and evaluate the first <code>global_step</code> or not.`,name:"first_step"},{anchor:"transformers.TrainingArguments.set_logging.nan_inf_filter",description:`<strong>nan_inf_filter</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to filter <code>nan</code> and <code>inf</code> losses for logging. If set to <code>True</code> the loss of every step that is | |
| <code>nan</code> or <code>inf</code> is filtered and the average loss of the current logging window is taken instead.</p> | |
| <div class="course-tip bg-gradient-to-br dark:bg-gradient-to-r before:border-green-500 dark:before:border-green-800 from-green-50 dark:from-gray-900 to-white dark:to-gray-950 border border-green-50 text-green-700 dark:text-gray-400"> | |
| <p><code>nan_inf_filter</code> only influences the logging of loss values, it does not change the behavior the | |
| gradient is computed or applied to the model.</p> | |
| </div>`,name:"nan_inf_filter"},{anchor:"transformers.TrainingArguments.set_logging.on_each_node",description:`<strong>on_each_node</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| In multinode distributed training, whether to log using <code>log_level</code> once per node, or only on the main | |
| node.`,name:"on_each_node"},{anchor:"transformers.TrainingArguments.set_logging.replica_level",description:`<strong>replica_level</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"passive"</code>) — | |
| Logger log level to use on replicas. Same choices as <code>log_level</code>`,name:"replica_level"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2754"}}),et=new be({props:{anchor:"transformers.TrainingArguments.set_logging.example",$$slots:{default:[Sm]},$$scope:{ctx:C}}}),xo=new x({props:{name:"set_lr_scheduler",anchor:"transformers.TrainingArguments.set_lr_scheduler",parameters:[{name:"name",val:": Union = 'linear'"},{name:"num_epochs",val:": float = 3.0"},{name:"max_steps",val:": int = -1"},{name:"warmup_ratio",val:": float = 0"},{name:"warmup_steps",val:": int = 0"}],parametersDescription:[{anchor:"transformers.TrainingArguments.set_lr_scheduler.name",description:`<strong>name</strong> (<code>str</code> or <code>SchedulerType</code>, <em>optional</em>, defaults to <code>"linear"</code>) — | |
| The scheduler type to use. See the documentation of <code>SchedulerType</code> for all possible values.`,name:"name"},{anchor:"transformers.TrainingArguments.set_lr_scheduler.num_epochs(float,",description:`<strong>num_epochs(<code>float</code>,</strong> <em>optional</em>, defaults to 3.0) — | |
| Total number of training epochs to perform (if not an integer, will perform the decimal part percents | |
| of the last epoch before stopping training).`,name:"num_epochs(float,"},{anchor:"transformers.TrainingArguments.set_lr_scheduler.max_steps",description:`<strong>max_steps</strong> (<code>int</code>, <em>optional</em>, defaults to -1) — | |
| If set to a positive number, the total number of training steps to perform. Overrides <code>num_train_epochs</code>. | |
| For a finite dataset, training is reiterated through the dataset (if all data is exhausted) until | |
| <code>max_steps</code> is reached.`,name:"max_steps"},{anchor:"transformers.TrainingArguments.set_lr_scheduler.warmup_ratio",description:`<strong>warmup_ratio</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| Ratio of total training steps used for a linear warmup from 0 to <code>learning_rate</code>.`,name:"warmup_ratio"},{anchor:"transformers.TrainingArguments.set_lr_scheduler.warmup_steps",description:`<strong>warmup_steps</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| Number of steps used for a linear warmup from 0 to <code>learning_rate</code>. Overrides any effect of | |
| <code>warmup_ratio</code>.`,name:"warmup_steps"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2950"}}),tt=new be({props:{anchor:"transformers.TrainingArguments.set_lr_scheduler.example",$$slots:{default:[Cm]},$$scope:{ctx:C}}}),qo=new x({props:{name:"set_optimizer",anchor:"transformers.TrainingArguments.set_optimizer",parameters:[{name:"name",val:": Union = 'adamw_torch'"},{name:"learning_rate",val:": float = 5e-05"},{name:"weight_decay",val:": float = 0"},{name:"beta1",val:": float = 0.9"},{name:"beta2",val:": float = 0.999"},{name:"epsilon",val:": float = 1e-08"},{name:"args",val:": Optional = None"}],parametersDescription:[{anchor:"transformers.TrainingArguments.set_optimizer.name",description:`<strong>name</strong> (<code>str</code> or <code>training_args.OptimizerNames</code>, <em>optional</em>, defaults to <code>"adamw_torch"</code>) — | |
| The optimizer to use: <code>"adamw_hf"</code>, <code>"adamw_torch"</code>, <code>"adamw_torch_fused"</code>, <code>"adamw_apex_fused"</code>, | |
| <code>"adamw_anyprecision"</code> or <code>"adafactor"</code>.`,name:"name"},{anchor:"transformers.TrainingArguments.set_optimizer.learning_rate",description:`<strong>learning_rate</strong> (<code>float</code>, <em>optional</em>, defaults to 5e-5) — | |
| The initial learning rate.`,name:"learning_rate"},{anchor:"transformers.TrainingArguments.set_optimizer.weight_decay",description:`<strong>weight_decay</strong> (<code>float</code>, <em>optional</em>, defaults to 0) — | |
| The weight decay to apply (if not zero) to all layers except all bias and LayerNorm weights.`,name:"weight_decay"},{anchor:"transformers.TrainingArguments.set_optimizer.beta1",description:`<strong>beta1</strong> (<code>float</code>, <em>optional</em>, defaults to 0.9) — | |
| The beta1 hyperparameter for the adam optimizer or its variants.`,name:"beta1"},{anchor:"transformers.TrainingArguments.set_optimizer.beta2",description:`<strong>beta2</strong> (<code>float</code>, <em>optional</em>, defaults to 0.999) — | |
| The beta2 hyperparameter for the adam optimizer or its variants.`,name:"beta2"},{anchor:"transformers.TrainingArguments.set_optimizer.epsilon",description:`<strong>epsilon</strong> (<code>float</code>, <em>optional</em>, defaults to 1e-8) — | |
| The epsilon hyperparameter for the adam optimizer or its variants.`,name:"epsilon"},{anchor:"transformers.TrainingArguments.set_optimizer.args",description:`<strong>args</strong> (<code>str</code>, <em>optional</em>) — | |
| Optional arguments that are supplied to AnyPrecisionAdamW (only useful when | |
| <code>optim="adamw_anyprecision"</code>).`,name:"args"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2899"}}),ot=new be({props:{anchor:"transformers.TrainingArguments.set_optimizer.example",$$slots:{default:[Pm]},$$scope:{ctx:C}}}),ko=new x({props:{name:"set_push_to_hub",anchor:"transformers.TrainingArguments.set_push_to_hub",parameters:[{name:"model_id",val:": str"},{name:"strategy",val:": Union = 'every_save'"},{name:"token",val:": Optional = None"},{name:"private_repo",val:": bool = False"},{name:"always_push",val:": bool = False"}],parametersDescription:[{anchor:"transformers.TrainingArguments.set_push_to_hub.model_id",description:`<strong>model_id</strong> (<code>str</code>) — | |
| The name of the repository to keep in sync with the local <em>output_dir</em>. It can be a simple model ID in | |
| which case the model will be pushed in your namespace. Otherwise it should be the whole repository | |
| name, for instance <code>"user_name/model"</code>, which allows you to push to an organization you are a member of | |
| with <code>"organization_name/model"</code>.`,name:"model_id"},{anchor:"transformers.TrainingArguments.set_push_to_hub.strategy",description:`<strong>strategy</strong> (<code>str</code> or <code>HubStrategy</code>, <em>optional</em>, defaults to <code>"every_save"</code>) — | |
| Defines the scope of what is pushed to the Hub and when. Possible values are:</p> | |
| <ul> | |
| <li><code>"end"</code>: push the model, its configuration, the processing_class e.g. tokenizer (if passed along to the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>) and a | |
| draft of a model card when the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer.save_model">save_model()</a> method is called.</li> | |
| <li><code>"every_save"</code>: push the model, its configuration, the processing_class e.g. tokenizer (if passed along to the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>) | |
| and | |
| a draft of a model card each time there is a model save. The pushes are asynchronous to not block | |
| training, and in case the save are very frequent, a new push is only attempted if the previous one is | |
| finished. A last push is made with the final model at the end of training.</li> | |
| <li><code>"checkpoint"</code>: like <code>"every_save"</code> but the latest checkpoint is also pushed in a subfolder named | |
| last-checkpoint, allowing you to resume training easily with | |
| <code>trainer.train(resume_from_checkpoint="last-checkpoint")</code>.</li> | |
| <li><code>"all_checkpoints"</code>: like <code>"checkpoint"</code> but all checkpoints are pushed like they appear in the | |
| output | |
| folder (so you will get one checkpoint folder per folder in your final repository)</li> | |
| </ul>`,name:"strategy"},{anchor:"transformers.TrainingArguments.set_push_to_hub.token",description:`<strong>token</strong> (<code>str</code>, <em>optional</em>) — | |
| The token to use to push the model to the Hub. Will default to the token in the cache folder obtained | |
| with <code>huggingface-cli login</code>.`,name:"token"},{anchor:"transformers.TrainingArguments.set_push_to_hub.private_repo",description:`<strong>private_repo</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| If True, the Hub repo will be set to private.`,name:"private_repo"},{anchor:"transformers.TrainingArguments.set_push_to_hub.always_push",description:`<strong>always_push</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Unless this is <code>True</code>, the <code>Trainer</code> will skip pushing a checkpoint when the previous push is not | |
| finished.`,name:"always_push"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2829"}}),nt=new Eo({props:{$$slots:{default:[Mm]},$$scope:{ctx:C}}}),rt=new be({props:{anchor:"transformers.TrainingArguments.set_push_to_hub.example",$$slots:{default:[Fm]},$$scope:{ctx:C}}}),$o=new x({props:{name:"set_save",anchor:"transformers.TrainingArguments.set_save",parameters:[{name:"strategy",val:": Union = 'steps'"},{name:"steps",val:": int = 500"},{name:"total_limit",val:": Optional = None"},{name:"on_each_node",val:": bool = False"}],parametersDescription:[{anchor:"transformers.TrainingArguments.set_save.strategy",description:`<strong>strategy</strong> (<code>str</code> or <code>IntervalStrategy</code>, <em>optional</em>, defaults to <code>"steps"</code>) — | |
| The checkpoint save strategy to adopt during training. Possible values are:</p> | |
| <ul> | |
| <li><code>"no"</code>: No save is done during training.</li> | |
| <li><code>"epoch"</code>: Save is done at the end of each epoch.</li> | |
| <li><code>"steps"</code>: Save is done every <code>save_steps</code>.</li> | |
| </ul>`,name:"strategy"},{anchor:"transformers.TrainingArguments.set_save.steps",description:`<strong>steps</strong> (<code>int</code>, <em>optional</em>, defaults to 500) — | |
| Number of updates steps before two checkpoint saves if <code>strategy="steps"</code>.`,name:"steps"},{anchor:"transformers.TrainingArguments.set_save.total_limit",description:`<strong>total_limit</strong> (<code>int</code>, <em>optional</em>) — | |
| If a value is passed, will limit the total amount of checkpoints. Deletes the older checkpoints in | |
| <code>output_dir</code>.`,name:"total_limit"},{anchor:"transformers.TrainingArguments.set_save.on_each_node",description:`<strong>on_each_node</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| When doing multi-node distributed training, whether to save models and checkpoints on each node, or | |
| only on the main one.</p> | |
| <p>This should not be activated when the different nodes use the same storage as the files will be saved | |
| with the same names for each node.`,name:"on_each_node"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2705"}}),at=new be({props:{anchor:"transformers.TrainingArguments.set_save.example",$$slots:{default:[zm]},$$scope:{ctx:C}}}),Ao=new x({props:{name:"set_testing",anchor:"transformers.TrainingArguments.set_testing",parameters:[{name:"batch_size",val:": int = 8"},{name:"loss_only",val:": bool = False"},{name:"jit_mode",val:": bool = False"}],parametersDescription:[{anchor:"transformers.TrainingArguments.set_testing.batch_size",description:`<strong>batch_size</strong> (<code>int</code> <em>optional</em>, defaults to 8) — | |
| The batch size per device (GPU/TPU core/CPU…) used for testing.`,name:"batch_size"},{anchor:"transformers.TrainingArguments.set_testing.loss_only",description:`<strong>loss_only</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Ignores all outputs except the loss.`,name:"loss_only"},{anchor:"transformers.TrainingArguments.set_testing.jit_mode",description:`<strong>jit_mode</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to use PyTorch jit trace for inference.`,name:"jit_mode"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2665"}}),st=new Eo({props:{$$slots:{default:[Dm]},$$scope:{ctx:C}}}),it=new be({props:{anchor:"transformers.TrainingArguments.set_testing.example",$$slots:{default:[Im]},$$scope:{ctx:C}}}),So=new x({props:{name:"set_training",anchor:"transformers.TrainingArguments.set_training",parameters:[{name:"learning_rate",val:": float = 5e-05"},{name:"batch_size",val:": int = 8"},{name:"weight_decay",val:": float = 0"},{name:"num_epochs",val:": float = 3"},{name:"max_steps",val:": int = -1"},{name:"gradient_accumulation_steps",val:": int = 1"},{name:"seed",val:": int = 42"},{name:"gradient_checkpointing",val:": bool = False"}],parametersDescription:[{anchor:"transformers.TrainingArguments.set_training.learning_rate",description:`<strong>learning_rate</strong> (<code>float</code>, <em>optional</em>, defaults to 5e-5) — | |
| The initial learning rate for the optimizer.`,name:"learning_rate"},{anchor:"transformers.TrainingArguments.set_training.batch_size",description:`<strong>batch_size</strong> (<code>int</code> <em>optional</em>, defaults to 8) — | |
| The batch size per device (GPU/TPU core/CPU…) used for training.`,name:"batch_size"},{anchor:"transformers.TrainingArguments.set_training.weight_decay",description:`<strong>weight_decay</strong> (<code>float</code>, <em>optional</em>, defaults to 0) — | |
| The weight decay to apply (if not zero) to all layers except all bias and LayerNorm weights in the | |
| optimizer.`,name:"weight_decay"},{anchor:"transformers.TrainingArguments.set_training.num_train_epochs(float,",description:`<strong>num_train_epochs(<code>float</code>,</strong> <em>optional</em>, defaults to 3.0) — | |
| Total number of training epochs to perform (if not an integer, will perform the decimal part percents | |
| of the last epoch before stopping training).`,name:"num_train_epochs(float,"},{anchor:"transformers.TrainingArguments.set_training.max_steps",description:`<strong>max_steps</strong> (<code>int</code>, <em>optional</em>, defaults to -1) — | |
| If set to a positive number, the total number of training steps to perform. Overrides <code>num_train_epochs</code>. | |
| For a finite dataset, training is reiterated through the dataset (if all data is exhausted) until | |
| <code>max_steps</code> is reached.`,name:"max_steps"},{anchor:"transformers.TrainingArguments.set_training.gradient_accumulation_steps",description:`<strong>gradient_accumulation_steps</strong> (<code>int</code>, <em>optional</em>, defaults to 1) — | |
| Number of updates steps to accumulate the gradients for, before performing a backward/update pass.</p> | |
| <div class="course-tip course-tip-orange bg-gradient-to-br dark:bg-gradient-to-r before:border-orange-500 dark:before:border-orange-800 from-orange-50 dark:from-gray-900 to-white dark:to-gray-950 border border-orange-50 text-orange-700 dark:text-gray-400"> | |
| <p>When using gradient accumulation, one step is counted as one step with backward pass. Therefore, | |
| logging, evaluation, save will be conducted every <code>gradient_accumulation_steps * xxx_step</code> training | |
| examples.</p> | |
| </div>`,name:"gradient_accumulation_steps"},{anchor:"transformers.TrainingArguments.set_training.seed",description:`<strong>seed</strong> (<code>int</code>, <em>optional</em>, defaults to 42) — | |
| Random seed that will be set at the beginning of training. To ensure reproducibility across runs, use | |
| the <code>~Trainer.model_init</code> function to instantiate the model if it has some randomly initialized | |
| parameters.`,name:"seed"},{anchor:"transformers.TrainingArguments.set_training.gradient_checkpointing",description:`<strong>gradient_checkpointing</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| If True, use gradient checkpointing to save memory at the expense of slower backward pass.`,name:"gradient_checkpointing"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2529"}}),lt=new Eo({props:{$$slots:{default:[Um]},$$scope:{ctx:C}}}),dt=new be({props:{anchor:"transformers.TrainingArguments.set_training.example",$$slots:{default:[Wm]},$$scope:{ctx:C}}}),Co=new x({props:{name:"to_dict",anchor:"transformers.TrainingArguments.to_dict",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2487"}}),Po=new x({props:{name:"to_json_string",anchor:"transformers.TrainingArguments.to_json_string",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2509"}}),Mo=new x({props:{name:"to_sanitized_dict",anchor:"transformers.TrainingArguments.to_sanitized_dict",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args.py#L2515"}}),Fo=new ua({props:{title:"Seq2SeqTrainingArguments",local:"transformers.Seq2SeqTrainingArguments ][ transformers.Seq2SeqTrainingArguments",headingTag:"h2"}}),zo=new x({props:{name:"class transformers.Seq2SeqTrainingArguments",anchor:"transformers.Seq2SeqTrainingArguments",parameters:[{name:"output_dir",val:": str"},{name:"overwrite_output_dir",val:": bool = False"},{name:"do_train",val:": bool = False"},{name:"do_eval",val:": bool = False"},{name:"do_predict",val:": bool = False"},{name:"eval_strategy",val:": Union = 'no'"},{name:"prediction_loss_only",val:": bool = False"},{name:"per_device_train_batch_size",val:": int = 8"},{name:"per_device_eval_batch_size",val:": int = 8"},{name:"per_gpu_train_batch_size",val:": Optional = None"},{name:"per_gpu_eval_batch_size",val:": Optional = None"},{name:"gradient_accumulation_steps",val:": int = 1"},{name:"eval_accumulation_steps",val:": Optional = None"},{name:"eval_delay",val:": Optional = 0"},{name:"torch_empty_cache_steps",val:": Optional = None"},{name:"learning_rate",val:": float = 5e-05"},{name:"weight_decay",val:": float = 0.0"},{name:"adam_beta1",val:": float = 0.9"},{name:"adam_beta2",val:": float = 0.999"},{name:"adam_epsilon",val:": float = 1e-08"},{name:"max_grad_norm",val:": float = 1.0"},{name:"num_train_epochs",val:": float = 3.0"},{name:"max_steps",val:": int = -1"},{name:"lr_scheduler_type",val:": Union = 'linear'"},{name:"lr_scheduler_kwargs",val:": Union = <factory>"},{name:"warmup_ratio",val:": float = 0.0"},{name:"warmup_steps",val:": int = 0"},{name:"log_level",val:": Optional = 'passive'"},{name:"log_level_replica",val:": Optional = 'warning'"},{name:"log_on_each_node",val:": bool = True"},{name:"logging_dir",val:": Optional = None"},{name:"logging_strategy",val:": Union = 'steps'"},{name:"logging_first_step",val:": bool = False"},{name:"logging_steps",val:": float = 500"},{name:"logging_nan_inf_filter",val:": bool = True"},{name:"save_strategy",val:": Union = 'steps'"},{name:"save_steps",val:": float = 500"},{name:"save_total_limit",val:": Optional = None"},{name:"save_safetensors",val:": Optional = True"},{name:"save_on_each_node",val:": bool = False"},{name:"save_only_model",val:": bool = False"},{name:"restore_callback_states_from_checkpoint",val:": bool = False"},{name:"no_cuda",val:": bool = False"},{name:"use_cpu",val:": bool = False"},{name:"use_mps_device",val:": bool = False"},{name:"seed",val:": int = 42"},{name:"data_seed",val:": Optional = None"},{name:"jit_mode_eval",val:": bool = False"},{name:"use_ipex",val:": bool = False"},{name:"bf16",val:": bool = False"},{name:"fp16",val:": bool = False"},{name:"fp16_opt_level",val:": str = 'O1'"},{name:"half_precision_backend",val:": str = 'auto'"},{name:"bf16_full_eval",val:": bool = False"},{name:"fp16_full_eval",val:": bool = False"},{name:"tf32",val:": Optional = None"},{name:"local_rank",val:": int = -1"},{name:"ddp_backend",val:": Optional = None"},{name:"tpu_num_cores",val:": Optional = None"},{name:"tpu_metrics_debug",val:": bool = False"},{name:"debug",val:": Union = ''"},{name:"dataloader_drop_last",val:": bool = False"},{name:"eval_steps",val:": Optional = None"},{name:"dataloader_num_workers",val:": int = 0"},{name:"dataloader_prefetch_factor",val:": Optional = None"},{name:"past_index",val:": int = -1"},{name:"run_name",val:": Optional = None"},{name:"disable_tqdm",val:": Optional = None"},{name:"remove_unused_columns",val:": Optional = True"},{name:"label_names",val:": Optional = None"},{name:"load_best_model_at_end",val:": Optional = False"},{name:"metric_for_best_model",val:": Optional = None"},{name:"greater_is_better",val:": Optional = None"},{name:"ignore_data_skip",val:": bool = False"},{name:"fsdp",val:": Union = ''"},{name:"fsdp_min_num_params",val:": int = 0"},{name:"fsdp_config",val:": Union = None"},{name:"fsdp_transformer_layer_cls_to_wrap",val:": Optional = None"},{name:"accelerator_config",val:": Union = None"},{name:"deepspeed",val:": Union = None"},{name:"label_smoothing_factor",val:": float = 0.0"},{name:"optim",val:": Union = 'adamw_torch'"},{name:"optim_args",val:": Optional = None"},{name:"adafactor",val:": bool = False"},{name:"group_by_length",val:": bool = False"},{name:"length_column_name",val:": Optional = 'length'"},{name:"report_to",val:": Union = None"},{name:"ddp_find_unused_parameters",val:": Optional = None"},{name:"ddp_bucket_cap_mb",val:": Optional = None"},{name:"ddp_broadcast_buffers",val:": Optional = None"},{name:"dataloader_pin_memory",val:": bool = True"},{name:"dataloader_persistent_workers",val:": bool = False"},{name:"skip_memory_metrics",val:": bool = True"},{name:"use_legacy_prediction_loop",val:": bool = False"},{name:"push_to_hub",val:": bool = False"},{name:"resume_from_checkpoint",val:": Optional = None"},{name:"hub_model_id",val:": Optional = None"},{name:"hub_strategy",val:": Union = 'every_save'"},{name:"hub_token",val:": Optional = None"},{name:"hub_private_repo",val:": bool = False"},{name:"hub_always_push",val:": bool = False"},{name:"gradient_checkpointing",val:": bool = False"},{name:"gradient_checkpointing_kwargs",val:": Union = None"},{name:"include_inputs_for_metrics",val:": bool = False"},{name:"include_for_metrics",val:": List = <factory>"},{name:"eval_do_concat_batches",val:": bool = True"},{name:"fp16_backend",val:": str = 'auto'"},{name:"evaluation_strategy",val:": Union = None"},{name:"push_to_hub_model_id",val:": Optional = None"},{name:"push_to_hub_organization",val:": Optional = None"},{name:"push_to_hub_token",val:": Optional = None"},{name:"mp_parameters",val:": str = ''"},{name:"auto_find_batch_size",val:": bool = False"},{name:"full_determinism",val:": bool = False"},{name:"torchdynamo",val:": Optional = None"},{name:"ray_scope",val:": Optional = 'last'"},{name:"ddp_timeout",val:": Optional = 1800"},{name:"torch_compile",val:": bool = False"},{name:"torch_compile_backend",val:": Optional = None"},{name:"torch_compile_mode",val:": Optional = None"},{name:"dispatch_batches",val:": Optional = None"},{name:"split_batches",val:": Optional = None"},{name:"include_tokens_per_second",val:": Optional = False"},{name:"include_num_input_tokens_seen",val:": Optional = False"},{name:"neftune_noise_alpha",val:": Optional = None"},{name:"optim_target_modules",val:": Union = None"},{name:"batch_eval_metrics",val:": bool = False"},{name:"eval_on_start",val:": bool = False"},{name:"use_liger_kernel",val:": Optional = False"},{name:"eval_use_gather_object",val:": Optional = False"},{name:"sortish_sampler",val:": bool = False"},{name:"predict_with_generate",val:": bool = False"},{name:"generation_max_length",val:": Optional = None"},{name:"generation_num_beams",val:": Optional = None"},{name:"generation_config",val:": Union = None"}],parametersDescription:[{anchor:"transformers.Seq2SeqTrainingArguments.output_dir",description:`<strong>output_dir</strong> (<code>str</code>) — | |
| The output directory where the model predictions and checkpoints will be written.`,name:"output_dir"},{anchor:"transformers.Seq2SeqTrainingArguments.overwrite_output_dir",description:`<strong>overwrite_output_dir</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| If <code>True</code>, overwrite the content of the output directory. Use this to continue training if <code>output_dir</code> | |
| points to a checkpoint directory.`,name:"overwrite_output_dir"},{anchor:"transformers.Seq2SeqTrainingArguments.do_train",description:`<strong>do_train</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to run training or not. This argument is not directly used by <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>, it’s intended to be used | |
| by your training/evaluation scripts instead. See the <a href="https://github.com/huggingface/transformers/tree/main/examples" rel="nofollow">example | |
| scripts</a> for more details.`,name:"do_train"},{anchor:"transformers.Seq2SeqTrainingArguments.do_eval",description:`<strong>do_eval</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether to run evaluation on the validation set or not. Will be set to <code>True</code> if <code>eval_strategy</code> is | |
| different from <code>"no"</code>. This argument is not directly used by <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>, it’s intended to be used by your | |
| training/evaluation scripts instead. See the <a href="https://github.com/huggingface/transformers/tree/main/examples" rel="nofollow">example | |
| scripts</a> for more details.`,name:"do_eval"},{anchor:"transformers.Seq2SeqTrainingArguments.do_predict",description:`<strong>do_predict</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to run predictions on the test set or not. This argument is not directly used by <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>, it’s | |
| intended to be used by your training/evaluation scripts instead. See the <a href="https://github.com/huggingface/transformers/tree/main/examples" rel="nofollow">example | |
| scripts</a> for more details.`,name:"do_predict"},{anchor:"transformers.Seq2SeqTrainingArguments.eval_strategy",description:`<strong>eval_strategy</strong> (<code>str</code> or <code>IntervalStrategy</code>, <em>optional</em>, defaults to <code>"no"</code>) — | |
| The evaluation strategy to adopt during training. Possible values are:</p> | |
| <ul> | |
| <li><code>"no"</code>: No evaluation is done during training.</li> | |
| <li><code>"steps"</code>: Evaluation is done (and logged) every <code>eval_steps</code>.</li> | |
| <li><code>"epoch"</code>: Evaluation is done at the end of each epoch.</li> | |
| </ul>`,name:"eval_strategy"},{anchor:"transformers.Seq2SeqTrainingArguments.prediction_loss_only",description:`<strong>prediction_loss_only</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| When performing evaluation and generating predictions, only returns the loss.`,name:"prediction_loss_only"},{anchor:"transformers.Seq2SeqTrainingArguments.per_device_train_batch_size",description:`<strong>per_device_train_batch_size</strong> (<code>int</code>, <em>optional</em>, defaults to 8) — | |
| The batch size per GPU/XPU/TPU/MPS/NPU core/CPU for training.`,name:"per_device_train_batch_size"},{anchor:"transformers.Seq2SeqTrainingArguments.per_device_eval_batch_size",description:`<strong>per_device_eval_batch_size</strong> (<code>int</code>, <em>optional</em>, defaults to 8) — | |
| The batch size per GPU/XPU/TPU/MPS/NPU core/CPU for evaluation.`,name:"per_device_eval_batch_size"},{anchor:"transformers.Seq2SeqTrainingArguments.gradient_accumulation_steps",description:`<strong>gradient_accumulation_steps</strong> (<code>int</code>, <em>optional</em>, defaults to 1) — | |
| Number of updates steps to accumulate the gradients for, before performing a backward/update pass.</p> | |
| <div class="course-tip course-tip-orange bg-gradient-to-br dark:bg-gradient-to-r before:border-orange-500 dark:before:border-orange-800 from-orange-50 dark:from-gray-900 to-white dark:to-gray-950 border border-orange-50 text-orange-700 dark:text-gray-400"> | |
| <p>When using gradient accumulation, one step is counted as one step with backward pass. Therefore, logging, | |
| evaluation, save will be conducted every <code>gradient_accumulation_steps * xxx_step</code> training examples.</p> | |
| </div>`,name:"gradient_accumulation_steps"},{anchor:"transformers.Seq2SeqTrainingArguments.eval_accumulation_steps",description:`<strong>eval_accumulation_steps</strong> (<code>int</code>, <em>optional</em>) — | |
| Number of predictions steps to accumulate the output tensors for, before moving the results to the CPU. If | |
| left unset, the whole predictions are accumulated on GPU/NPU/TPU before being moved to the CPU (faster but | |
| requires more memory).`,name:"eval_accumulation_steps"},{anchor:"transformers.Seq2SeqTrainingArguments.eval_delay",description:`<strong>eval_delay</strong> (<code>float</code>, <em>optional</em>) — | |
| Number of epochs or steps to wait for before the first evaluation can be performed, depending on the | |
| eval_strategy.`,name:"eval_delay"},{anchor:"transformers.Seq2SeqTrainingArguments.torch_empty_cache_steps",description:`<strong>torch_empty_cache_steps</strong> (<code>int</code>, <em>optional</em>) — | |
| Number of steps to wait before calling <code>torch.<device>.empty_cache()</code>. If left unset or set to None, cache will not be emptied.</p> | |
| <div class="course-tip bg-gradient-to-br dark:bg-gradient-to-r before:border-green-500 dark:before:border-green-800 from-green-50 dark:from-gray-900 to-white dark:to-gray-950 border border-green-50 text-green-700 dark:text-gray-400"> | |
| <p>This can help avoid CUDA out-of-memory errors by lowering peak VRAM usage at a cost of about <a href="https://github.com/huggingface/transformers/issues/31372" rel="nofollow">10% slower performance</a>.</p> | |
| </div>`,name:"torch_empty_cache_steps"},{anchor:"transformers.Seq2SeqTrainingArguments.learning_rate",description:`<strong>learning_rate</strong> (<code>float</code>, <em>optional</em>, defaults to 5e-5) — | |
| The initial learning rate for <code>AdamW</code> optimizer.`,name:"learning_rate"},{anchor:"transformers.Seq2SeqTrainingArguments.weight_decay",description:`<strong>weight_decay</strong> (<code>float</code>, <em>optional</em>, defaults to 0) — | |
| The weight decay to apply (if not zero) to all layers except all bias and LayerNorm weights in <code>AdamW</code> | |
| optimizer.`,name:"weight_decay"},{anchor:"transformers.Seq2SeqTrainingArguments.adam_beta1",description:`<strong>adam_beta1</strong> (<code>float</code>, <em>optional</em>, defaults to 0.9) — | |
| The beta1 hyperparameter for the <code>AdamW</code> optimizer.`,name:"adam_beta1"},{anchor:"transformers.Seq2SeqTrainingArguments.adam_beta2",description:`<strong>adam_beta2</strong> (<code>float</code>, <em>optional</em>, defaults to 0.999) — | |
| The beta2 hyperparameter for the <code>AdamW</code> optimizer.`,name:"adam_beta2"},{anchor:"transformers.Seq2SeqTrainingArguments.adam_epsilon",description:`<strong>adam_epsilon</strong> (<code>float</code>, <em>optional</em>, defaults to 1e-8) — | |
| The epsilon hyperparameter for the <code>AdamW</code> optimizer.`,name:"adam_epsilon"},{anchor:"transformers.Seq2SeqTrainingArguments.max_grad_norm",description:`<strong>max_grad_norm</strong> (<code>float</code>, <em>optional</em>, defaults to 1.0) — | |
| Maximum gradient norm (for gradient clipping).`,name:"max_grad_norm"},{anchor:"transformers.Seq2SeqTrainingArguments.num_train_epochs(float,",description:`<strong>num_train_epochs(<code>float</code>,</strong> <em>optional</em>, defaults to 3.0) — | |
| Total number of training epochs to perform (if not an integer, will perform the decimal part percents of | |
| the last epoch before stopping training).`,name:"num_train_epochs(float,"},{anchor:"transformers.Seq2SeqTrainingArguments.max_steps",description:`<strong>max_steps</strong> (<code>int</code>, <em>optional</em>, defaults to -1) — | |
| If set to a positive number, the total number of training steps to perform. Overrides <code>num_train_epochs</code>. | |
| For a finite dataset, training is reiterated through the dataset (if all data is exhausted) until | |
| <code>max_steps</code> is reached.`,name:"max_steps"},{anchor:"transformers.Seq2SeqTrainingArguments.lr_scheduler_type",description:`<strong>lr_scheduler_type</strong> (<code>str</code> or <code>SchedulerType</code>, <em>optional</em>, defaults to <code>"linear"</code>) — | |
| The scheduler type to use. See the documentation of <code>SchedulerType</code> for all possible values.`,name:"lr_scheduler_type"},{anchor:"transformers.Seq2SeqTrainingArguments.lr_scheduler_kwargs",description:`<strong>lr_scheduler_kwargs</strong> (‘dict’, <em>optional</em>, defaults to {}) — | |
| The extra arguments for the lr_scheduler. See the documentation of each scheduler for possible values.`,name:"lr_scheduler_kwargs"},{anchor:"transformers.Seq2SeqTrainingArguments.warmup_ratio",description:`<strong>warmup_ratio</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| Ratio of total training steps used for a linear warmup from 0 to <code>learning_rate</code>.`,name:"warmup_ratio"},{anchor:"transformers.Seq2SeqTrainingArguments.warmup_steps",description:`<strong>warmup_steps</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| Number of steps used for a linear warmup from 0 to <code>learning_rate</code>. Overrides any effect of <code>warmup_ratio</code>.`,name:"warmup_steps"},{anchor:"transformers.Seq2SeqTrainingArguments.log_level",description:`<strong>log_level</strong> (<code>str</code>, <em>optional</em>, defaults to <code>passive</code>) — | |
| Logger log level to use on the main process. Possible choices are the log levels as strings: ‘debug’, | |
| ‘info’, ‘warning’, ‘error’ and ‘critical’, plus a ‘passive’ level which doesn’t set anything and keeps the | |
| current log level for the Transformers library (which will be <code>"warning"</code> by default).`,name:"log_level"},{anchor:"transformers.Seq2SeqTrainingArguments.log_level_replica",description:`<strong>log_level_replica</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"warning"</code>) — | |
| Logger log level to use on replicas. Same choices as <code>log_level</code>”`,name:"log_level_replica"},{anchor:"transformers.Seq2SeqTrainingArguments.log_on_each_node",description:`<strong>log_on_each_node</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| In multinode distributed training, whether to log using <code>log_level</code> once per node, or only on the main | |
| node.`,name:"log_on_each_node"},{anchor:"transformers.Seq2SeqTrainingArguments.logging_dir",description:`<strong>logging_dir</strong> (<code>str</code>, <em>optional</em>) — | |
| <a href="https://www.tensorflow.org/tensorboard" rel="nofollow">TensorBoard</a> log directory. Will default to | |
| *output_dir/runs/<strong>CURRENT_DATETIME_HOSTNAME*</strong>.`,name:"logging_dir"},{anchor:"transformers.Seq2SeqTrainingArguments.logging_strategy",description:`<strong>logging_strategy</strong> (<code>str</code> or <code>IntervalStrategy</code>, <em>optional</em>, defaults to <code>"steps"</code>) — | |
| The logging strategy to adopt during training. Possible values are:</p> | |
| <ul> | |
| <li><code>"no"</code>: No logging is done during training.</li> | |
| <li><code>"epoch"</code>: Logging is done at the end of each epoch.</li> | |
| <li><code>"steps"</code>: Logging is done every <code>logging_steps</code>.</li> | |
| </ul>`,name:"logging_strategy"},{anchor:"transformers.Seq2SeqTrainingArguments.logging_first_step",description:`<strong>logging_first_step</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to log the first <code>global_step</code> or not.`,name:"logging_first_step"},{anchor:"transformers.Seq2SeqTrainingArguments.logging_steps",description:`<strong>logging_steps</strong> (<code>int</code> or <code>float</code>, <em>optional</em>, defaults to 500) — | |
| Number of update steps between two logs if <code>logging_strategy="steps"</code>. Should be an integer or a float in | |
| range <code>[0,1)</code>. If smaller than 1, will be interpreted as ratio of total training steps.`,name:"logging_steps"},{anchor:"transformers.Seq2SeqTrainingArguments.logging_nan_inf_filter",description:`<strong>logging_nan_inf_filter</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to filter <code>nan</code> and <code>inf</code> losses for logging. If set to <code>True</code> the loss of every step that is <code>nan</code> | |
| or <code>inf</code> is filtered and the average loss of the current logging window is taken instead.</p> | |
| <div class="course-tip bg-gradient-to-br dark:bg-gradient-to-r before:border-green-500 dark:before:border-green-800 from-green-50 dark:from-gray-900 to-white dark:to-gray-950 border border-green-50 text-green-700 dark:text-gray-400"> | |
| <p><code>logging_nan_inf_filter</code> only influences the logging of loss values, it does not change the behavior the | |
| gradient is computed or applied to the model.</p> | |
| </div>`,name:"logging_nan_inf_filter"},{anchor:"transformers.Seq2SeqTrainingArguments.save_strategy",description:`<strong>save_strategy</strong> (<code>str</code> or <code>IntervalStrategy</code>, <em>optional</em>, defaults to <code>"steps"</code>) — | |
| The checkpoint save strategy to adopt during training. Possible values are:</p> | |
| <ul> | |
| <li><code>"no"</code>: No save is done during training.</li> | |
| <li><code>"epoch"</code>: Save is done at the end of each epoch.</li> | |
| <li><code>"steps"</code>: Save is done every <code>save_steps</code>.</li> | |
| </ul> | |
| <p>If <code>"epoch"</code> or <code>"steps"</code> is chosen, saving will also be performed at the | |
| very end of training, always.`,name:"save_strategy"},{anchor:"transformers.Seq2SeqTrainingArguments.save_steps",description:`<strong>save_steps</strong> (<code>int</code> or <code>float</code>, <em>optional</em>, defaults to 500) — | |
| Number of updates steps before two checkpoint saves if <code>save_strategy="steps"</code>. Should be an integer or a | |
| float in range <code>[0,1)</code>. If smaller than 1, will be interpreted as ratio of total training steps.`,name:"save_steps"},{anchor:"transformers.Seq2SeqTrainingArguments.save_total_limit",description:`<strong>save_total_limit</strong> (<code>int</code>, <em>optional</em>) — | |
| If a value is passed, will limit the total amount of checkpoints. Deletes the older checkpoints in | |
| <code>output_dir</code>. When <code>load_best_model_at_end</code> is enabled, the “best” checkpoint according to | |
| <code>metric_for_best_model</code> will always be retained in addition to the most recent ones. For example, for | |
| <code>save_total_limit=5</code> and <code>load_best_model_at_end</code>, the four last checkpoints will always be retained | |
| alongside the best model. When <code>save_total_limit=1</code> and <code>load_best_model_at_end</code>, it is possible that two | |
| checkpoints are saved: the last one and the best one (if they are different).`,name:"save_total_limit"},{anchor:"transformers.Seq2SeqTrainingArguments.save_safetensors",description:`<strong>save_safetensors</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Use <a href="https://huggingface.co/docs/safetensors" rel="nofollow">safetensors</a> saving and loading for state dicts instead of | |
| default <code>torch.load</code> and <code>torch.save</code>.`,name:"save_safetensors"},{anchor:"transformers.Seq2SeqTrainingArguments.save_on_each_node",description:`<strong>save_on_each_node</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| When doing multi-node distributed training, whether to save models and checkpoints on each node, or only on | |
| the main one.</p> | |
| <p>This should not be activated when the different nodes use the same storage as the files will be saved with | |
| the same names for each node.`,name:"save_on_each_node"},{anchor:"transformers.Seq2SeqTrainingArguments.save_only_model",description:`<strong>save_only_model</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| When checkpointing, whether to only save the model, or also the optimizer, scheduler & rng state. | |
| Note that when this is true, you won’t be able to resume training from checkpoint. | |
| This enables you to save storage by not storing the optimizer, scheduler & rng state. | |
| You can only load the model using <code>from_pretrained</code> with this option set to <code>True</code>.`,name:"save_only_model"},{anchor:"transformers.Seq2SeqTrainingArguments.restore_callback_states_from_checkpoint",description:`<strong>restore_callback_states_from_checkpoint</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to restore the callback states from the checkpoint. If <code>True</code>, will override | |
| callbacks passed to the <code>Trainer</code> if they exist in the checkpoint.”`,name:"restore_callback_states_from_checkpoint"},{anchor:"transformers.Seq2SeqTrainingArguments.use_cpu",description:`<strong>use_cpu</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to use cpu. If set to False, we will use cuda or mps device if available.`,name:"use_cpu"},{anchor:"transformers.Seq2SeqTrainingArguments.seed",description:`<strong>seed</strong> (<code>int</code>, <em>optional</em>, defaults to 42) — | |
| Random seed that will be set at the beginning of training. To ensure reproducibility across runs, use the | |
| <code>~Trainer.model_init</code> function to instantiate the model if it has some randomly initialized parameters.`,name:"seed"},{anchor:"transformers.Seq2SeqTrainingArguments.data_seed",description:`<strong>data_seed</strong> (<code>int</code>, <em>optional</em>) — | |
| Random seed to be used with data samplers. If not set, random generators for data sampling will use the | |
| same seed as <code>seed</code>. This can be used to ensure reproducibility of data sampling, independent of the model | |
| seed.`,name:"data_seed"},{anchor:"transformers.Seq2SeqTrainingArguments.jit_mode_eval",description:`<strong>jit_mode_eval</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to use PyTorch jit trace for inference.`,name:"jit_mode_eval"},{anchor:"transformers.Seq2SeqTrainingArguments.use_ipex",description:`<strong>use_ipex</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Use Intel extension for PyTorch when it is available. <a href="https://github.com/intel/intel-extension-for-pytorch" rel="nofollow">IPEX | |
| installation</a>.`,name:"use_ipex"},{anchor:"transformers.Seq2SeqTrainingArguments.bf16",description:`<strong>bf16</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to use bf16 16-bit (mixed) precision training instead of 32-bit training. Requires Ampere or higher | |
| NVIDIA architecture or using CPU (use_cpu) or Ascend NPU. This is an experimental API and it may change.`,name:"bf16"},{anchor:"transformers.Seq2SeqTrainingArguments.fp16",description:`<strong>fp16</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to use fp16 16-bit (mixed) precision training instead of 32-bit training.`,name:"fp16"},{anchor:"transformers.Seq2SeqTrainingArguments.fp16_opt_level",description:`<strong>fp16_opt_level</strong> (<code>str</code>, <em>optional</em>, defaults to ‘O1’) — | |
| For <code>fp16</code> training, Apex AMP optimization level selected in [‘O0’, ‘O1’, ‘O2’, and ‘O3’]. See details on | |
| the <a href="https://nvidia.github.io/apex/amp" rel="nofollow">Apex documentation</a>.`,name:"fp16_opt_level"},{anchor:"transformers.Seq2SeqTrainingArguments.fp16_backend",description:`<strong>fp16_backend</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"auto"</code>) — | |
| This argument is deprecated. Use <code>half_precision_backend</code> instead.`,name:"fp16_backend"},{anchor:"transformers.Seq2SeqTrainingArguments.half_precision_backend",description:`<strong>half_precision_backend</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"auto"</code>) — | |
| The backend to use for mixed precision training. Must be one of <code>"auto", "apex", "cpu_amp"</code>. <code>"auto"</code> will | |
| use CPU/CUDA AMP or APEX depending on the PyTorch version detected, while the other choices will force the | |
| requested backend.`,name:"half_precision_backend"},{anchor:"transformers.Seq2SeqTrainingArguments.bf16_full_eval",description:`<strong>bf16_full_eval</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to use full bfloat16 evaluation instead of 32-bit. This will be faster and save memory but can harm | |
| metric values. This is an experimental API and it may change.`,name:"bf16_full_eval"},{anchor:"transformers.Seq2SeqTrainingArguments.fp16_full_eval",description:`<strong>fp16_full_eval</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to use full float16 evaluation instead of 32-bit. This will be faster and save memory but can harm | |
| metric values.`,name:"fp16_full_eval"},{anchor:"transformers.Seq2SeqTrainingArguments.tf32",description:`<strong>tf32</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether to enable the TF32 mode, available in Ampere and newer GPU architectures. The default value depends | |
| on PyTorch’s version default of <code>torch.backends.cuda.matmul.allow_tf32</code>. For more details please refer to | |
| the <a href="https://huggingface.co/docs/transformers/perf_train_gpu_one#tf32" rel="nofollow">TF32</a> documentation. This is an | |
| experimental API and it may change.`,name:"tf32"},{anchor:"transformers.Seq2SeqTrainingArguments.local_rank",description:`<strong>local_rank</strong> (<code>int</code>, <em>optional</em>, defaults to -1) — | |
| Rank of the process during distributed training.`,name:"local_rank"},{anchor:"transformers.Seq2SeqTrainingArguments.ddp_backend",description:`<strong>ddp_backend</strong> (<code>str</code>, <em>optional</em>) — | |
| The backend to use for distributed training. Must be one of <code>"nccl"</code>, <code>"mpi"</code>, <code>"ccl"</code>, <code>"gloo"</code>, <code>"hccl"</code>.`,name:"ddp_backend"},{anchor:"transformers.Seq2SeqTrainingArguments.tpu_num_cores",description:`<strong>tpu_num_cores</strong> (<code>int</code>, <em>optional</em>) — | |
| When training on TPU, the number of TPU cores (automatically passed by launcher script).`,name:"tpu_num_cores"},{anchor:"transformers.Seq2SeqTrainingArguments.dataloader_drop_last",description:`<strong>dataloader_drop_last</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to drop the last incomplete batch (if the length of the dataset is not divisible by the batch size) | |
| or not.`,name:"dataloader_drop_last"},{anchor:"transformers.Seq2SeqTrainingArguments.eval_steps",description:`<strong>eval_steps</strong> (<code>int</code> or <code>float</code>, <em>optional</em>) — | |
| Number of update steps between two evaluations if <code>eval_strategy="steps"</code>. Will default to the same | |
| value as <code>logging_steps</code> if not set. Should be an integer or a float in range <code>[0,1)</code>. If smaller than 1, | |
| will be interpreted as ratio of total training steps.`,name:"eval_steps"},{anchor:"transformers.Seq2SeqTrainingArguments.dataloader_num_workers",description:`<strong>dataloader_num_workers</strong> (<code>int</code>, <em>optional</em>, defaults to 0) — | |
| Number of subprocesses to use for data loading (PyTorch only). 0 means that the data will be loaded in the | |
| main process.`,name:"dataloader_num_workers"},{anchor:"transformers.Seq2SeqTrainingArguments.past_index",description:`<strong>past_index</strong> (<code>int</code>, <em>optional</em>, defaults to -1) — | |
| Some models like <a href="../model_doc/transformerxl">TransformerXL</a> or <a href="../model_doc/xlnet">XLNet</a> can make use of | |
| the past hidden states for their predictions. If this argument is set to a positive int, the <code>Trainer</code> will | |
| use the corresponding output (usually index 2) as the past state and feed it to the model at the next | |
| training step under the keyword argument <code>mems</code>.`,name:"past_index"},{anchor:"transformers.Seq2SeqTrainingArguments.run_name",description:`<strong>run_name</strong> (<code>str</code>, <em>optional</em>, defaults to <code>output_dir</code>) — | |
| A descriptor for the run. Typically used for <a href="https://www.wandb.com/" rel="nofollow">wandb</a>, | |
| <a href="https://www.mlflow.org/" rel="nofollow">mlflow</a> and <a href="https://www.comet.com/site" rel="nofollow">comet</a> logging. If not specified, will | |
| be the same as <code>output_dir</code>.`,name:"run_name"},{anchor:"transformers.Seq2SeqTrainingArguments.disable_tqdm",description:`<strong>disable_tqdm</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to disable the tqdm progress bars and table of metrics produced by | |
| <code>~notebook.NotebookTrainingTracker</code> in Jupyter Notebooks. Will default to <code>True</code> if the logging level is | |
| set to warn or lower (default), <code>False</code> otherwise.`,name:"disable_tqdm"},{anchor:"transformers.Seq2SeqTrainingArguments.remove_unused_columns",description:`<strong>remove_unused_columns</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether or not to automatically remove the columns unused by the model forward method.`,name:"remove_unused_columns"},{anchor:"transformers.Seq2SeqTrainingArguments.label_names",description:`<strong>label_names</strong> (<code>List[str]</code>, <em>optional</em>) — | |
| The list of keys in your dictionary of inputs that correspond to the labels.</p> | |
| <p>Will eventually default to the list of argument names accepted by the model that contain the word “label”, | |
| except if the model used is one of the <code>XxxForQuestionAnswering</code> in which case it will also include the | |
| <code>["start_positions", "end_positions"]</code> keys.`,name:"label_names"},{anchor:"transformers.Seq2SeqTrainingArguments.load_best_model_at_end",description:`<strong>load_best_model_at_end</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to load the best model found during training at the end of training. When this option is | |
| enabled, the best checkpoint will always be saved. See | |
| <a href="https://huggingface.co/docs/transformers/main_classes/trainer#transformers.TrainingArguments.save_total_limit" rel="nofollow"><code>save_total_limit</code></a> | |
| for more.</p> | |
| <div class="course-tip bg-gradient-to-br dark:bg-gradient-to-r before:border-green-500 dark:before:border-green-800 from-green-50 dark:from-gray-900 to-white dark:to-gray-950 border border-green-50 text-green-700 dark:text-gray-400"> | |
| <p>When set to <code>True</code>, the parameters <code>save_strategy</code> needs to be the same as <code>eval_strategy</code>, and in | |
| the case it is “steps”, <code>save_steps</code> must be a round multiple of <code>eval_steps</code>.</p> | |
| </div>`,name:"load_best_model_at_end"},{anchor:"transformers.Seq2SeqTrainingArguments.metric_for_best_model",description:`<strong>metric_for_best_model</strong> (<code>str</code>, <em>optional</em>) — | |
| Use in conjunction with <code>load_best_model_at_end</code> to specify the metric to use to compare two different | |
| models. Must be the name of a metric returned by the evaluation with or without the prefix <code>"eval_"</code>. Will | |
| default to <code>"loss"</code> if unspecified and <code>load_best_model_at_end=True</code> (to use the evaluation loss).</p> | |
| <p>If you set this value, <code>greater_is_better</code> will default to <code>True</code>. Don’t forget to set it to <code>False</code> if | |
| your metric is better when lower.`,name:"metric_for_best_model"},{anchor:"transformers.Seq2SeqTrainingArguments.greater_is_better",description:`<strong>greater_is_better</strong> (<code>bool</code>, <em>optional</em>) — | |
| Use in conjunction with <code>load_best_model_at_end</code> and <code>metric_for_best_model</code> to specify if better models | |
| should have a greater metric or not. Will default to:</p> | |
| <ul> | |
| <li><code>True</code> if <code>metric_for_best_model</code> is set to a value that doesn’t end in <code>"loss"</code>.</li> | |
| <li><code>False</code> if <code>metric_for_best_model</code> is not set, or set to a value that ends in <code>"loss"</code>.</li> | |
| </ul>`,name:"greater_is_better"},{anchor:"transformers.Seq2SeqTrainingArguments.ignore_data_skip",description:`<strong>ignore_data_skip</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| When resuming training, whether or not to skip the epochs and batches to get the data loading at the same | |
| stage as in the previous training. If set to <code>True</code>, the training will begin faster (as that skipping step | |
| can take a long time) but will not yield the same results as the interrupted training would have.`,name:"ignore_data_skip"},{anchor:"transformers.Seq2SeqTrainingArguments.fsdp",description:`<strong>fsdp</strong> (<code>bool</code>, <code>str</code> or list of <code>FSDPOption</code>, <em>optional</em>, defaults to <code>''</code>) — | |
| Use PyTorch Distributed Parallel Training (in distributed training only).</p> | |
| <p>A list of options along the following:</p> | |
| <ul> | |
| <li><code>"full_shard"</code>: Shard parameters, gradients and optimizer states.</li> | |
| <li><code>"shard_grad_op"</code>: Shard optimizer states and gradients.</li> | |
| <li><code>"hybrid_shard"</code>: Apply <code>FULL_SHARD</code> within a node, and replicate parameters across nodes.</li> | |
| <li><code>"hybrid_shard_zero2"</code>: Apply <code>SHARD_GRAD_OP</code> within a node, and replicate parameters across nodes.</li> | |
| <li><code>"offload"</code>: Offload parameters and gradients to CPUs (only compatible with <code>"full_shard"</code> and | |
| <code>"shard_grad_op"</code>).</li> | |
| <li><code>"auto_wrap"</code>: Automatically recursively wrap layers with FSDP using <code>default_auto_wrap_policy</code>.</li> | |
| </ul>`,name:"fsdp"},{anchor:"transformers.Seq2SeqTrainingArguments.fsdp_config",description:`<strong>fsdp_config</strong> (<code>str</code> or <code>dict</code>, <em>optional</em>) — | |
| Config to be used with fsdp (Pytorch Distributed Parallel Training). The value is either a location of | |
| fsdp json config file (e.g., <code>fsdp_config.json</code>) or an already loaded json file as <code>dict</code>.</p> | |
| <p>A List of config and its options:</p> | |
| <ul> | |
| <li> | |
| <p>min_num_params (<code>int</code>, <em>optional</em>, defaults to <code>0</code>): | |
| FSDP’s minimum number of parameters for Default Auto Wrapping. (useful only when <code>fsdp</code> field is | |
| passed).</p> | |
| </li> | |
| <li> | |
| <p>transformer_layer_cls_to_wrap (<code>List[str]</code>, <em>optional</em>): | |
| List of transformer layer class names (case-sensitive) to wrap, e.g, <code>BertLayer</code>, <code>GPTJBlock</code>, | |
| <code>T5Block</code> … (useful only when <code>fsdp</code> flag is passed).</p> | |
| </li> | |
| <li> | |
| <p>backward_prefetch (<code>str</code>, <em>optional</em>) | |
| FSDP’s backward prefetch mode. Controls when to prefetch next set of parameters (useful only when | |
| <code>fsdp</code> field is passed).</p> | |
| <p>A list of options along the following:</p> | |
| <ul> | |
| <li><code>"backward_pre"</code> : Prefetches the next set of parameters before the current set of parameter’s | |
| gradient | |
| computation.</li> | |
| <li><code>"backward_post"</code> : This prefetches the next set of parameters after the current set of | |
| parameter’s | |
| gradient computation.</li> | |
| </ul> | |
| </li> | |
| <li> | |
| <p>forward_prefetch (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) | |
| FSDP’s forward prefetch mode (useful only when <code>fsdp</code> field is passed). | |
| If <code>"True"</code>, then FSDP explicitly prefetches the next upcoming all-gather while executing in the | |
| forward pass.</p> | |
| </li> | |
| <li> | |
| <p>limit_all_gathers (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) | |
| FSDP’s limit_all_gathers (useful only when <code>fsdp</code> field is passed). | |
| If <code>"True"</code>, FSDP explicitly synchronizes the CPU thread to prevent too many in-flight | |
| all-gathers.</p> | |
| </li> | |
| <li> | |
| <p>use_orig_params (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) | |
| If <code>"True"</code>, allows non-uniform <code>requires_grad</code> during init, which means support for interspersed | |
| frozen and trainable paramteres. Useful in cases such as parameter-efficient fine-tuning. Please | |
| refer this | |
| [blog](<a href="https://dev-discuss.pytorch.org/t/rethinking-pytorch-fully-sharded-data-parallel-fsdp-from-first-principles/1019" rel="nofollow">https://dev-discuss.pytorch.org/t/rethinking-pytorch-fully-sharded-data-parallel-fsdp-from-first-principles/1019</a></p> | |
| </li> | |
| <li> | |
| <p>sync_module_states (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) | |
| If <code>"True"</code>, each individually wrapped FSDP unit will broadcast module parameters from rank 0 to | |
| ensure they are the same across all ranks after initialization</p> | |
| </li> | |
| <li> | |
| <p>cpu_ram_efficient_loading (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) | |
| If <code>"True"</code>, only the first process loads the pretrained model checkpoint while all other processes | |
| have empty weights. When this setting as <code>"True"</code>, <code>sync_module_states</code> also must to be <code>"True"</code>, | |
| otherwise all the processes except the main process would have random weights leading to unexpected | |
| behaviour during training.</p> | |
| </li> | |
| <li> | |
| <p>activation_checkpointing (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>): | |
| If <code>"True"</code>, activation checkpointing is a technique to reduce memory usage by clearing activations of | |
| certain layers and recomputing them during a backward pass. Effectively, this trades extra | |
| computation time for reduced memory usage.</p> | |
| </li> | |
| <li> | |
| <p>xla (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>): | |
| Whether to use PyTorch/XLA Fully Sharded Data Parallel Training. This is an experimental feature | |
| and its API may evolve in the future.</p> | |
| </li> | |
| <li> | |
| <p>xla_fsdp_settings (<code>dict</code>, <em>optional</em>) | |
| The value is a dictionary which stores the XLA FSDP wrapping parameters.</p> | |
| <p>For a complete list of options, please see <a href="https://github.com/pytorch/xla/blob/master/torch_xla/distributed/fsdp/xla_fully_sharded_data_parallel.py" rel="nofollow">here</a>.</p> | |
| </li> | |
| <li> | |
| <p>xla_fsdp_grad_ckpt (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>): | |
| Will use gradient checkpointing over each nested XLA FSDP wrapped layer. This setting can only be | |
| used when the xla flag is set to true, and an auto wrapping policy is specified through | |
| fsdp_min_num_params or fsdp_transformer_layer_cls_to_wrap.</p> | |
| </li> | |
| </ul>`,name:"fsdp_config"},{anchor:"transformers.Seq2SeqTrainingArguments.deepspeed",description:`<strong>deepspeed</strong> (<code>str</code> or <code>dict</code>, <em>optional</em>) — | |
| Use <a href="https://github.com/microsoft/deepspeed" rel="nofollow">Deepspeed</a>. This is an experimental feature and its API may | |
| evolve in the future. The value is either the location of DeepSpeed json config file (e.g., | |
| <code>ds_config.json</code>) or an already loaded json file as a <code>dict</code>”</p> | |
| <div class="course-tip course-tip-orange bg-gradient-to-br dark:bg-gradient-to-r before:border-orange-500 dark:before:border-orange-800 from-orange-50 dark:from-gray-900 to-white dark:to-gray-950 border border-orange-50 text-orange-700 dark:text-gray-400"> | |
| If enabling any Zero-init, make sure that your model is not initialized until | |
| *after* initializing the \`TrainingArguments\`, else it will not be applied. | |
| </div>`,name:"deepspeed"},{anchor:"transformers.Seq2SeqTrainingArguments.accelerator_config",description:`<strong>accelerator_config</strong> (<code>str</code>, <code>dict</code>, or <code>AcceleratorConfig</code>, <em>optional</em>) — | |
| Config to be used with the internal <code>Accelerator</code> implementation. The value is either a location of | |
| accelerator json config file (e.g., <code>accelerator_config.json</code>), an already loaded json file as <code>dict</code>, | |
| or an instance of <code>AcceleratorConfig</code>.</p> | |
| <p>A list of config and its options:</p> | |
| <ul> | |
| <li>split_batches (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>): | |
| Whether or not the accelerator should split the batches yielded by the dataloaders across the devices. If | |
| <code>True</code> the actual batch size used will be the same on any kind of distributed processes, but it must be a | |
| round multiple of the <code>num_processes</code> you are using. If <code>False</code>, actual batch size used will be the one set | |
| in your script multiplied by the number of processes.</li> | |
| <li>dispatch_batches (<code>bool</code>, <em>optional</em>): | |
| If set to <code>True</code>, the dataloader prepared by the Accelerator is only iterated through on the main process | |
| and then the batches are split and broadcast to each process. Will default to <code>True</code> for <code>DataLoader</code> whose | |
| underlying dataset is an <code>IterableDataset</code>, <code>False</code> otherwise.</li> | |
| <li>even_batches (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>): | |
| If set to <code>True</code>, in cases where the total batch size across all processes does not exactly divide the | |
| dataset, samples at the start of the dataset will be duplicated so the batch can be divided equally among | |
| all workers.</li> | |
| <li>use_seedable_sampler (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>): | |
| Whether or not use a fully seedable random sampler (<code>accelerate.data_loader.SeedableRandomSampler</code>). Ensures | |
| training results are fully reproducable using a different sampling technique. While seed-to-seed results | |
| may differ, on average the differences are neglible when using multiple different seeds to compare. Should | |
| also be ran with <code>~utils.set_seed</code> for the best results.</li> | |
| <li>use_configured_state (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>): | |
| Whether or not to use a pre-configured <code>AcceleratorState</code> or <code>PartialState</code> defined before calling <code>TrainingArguments</code>. | |
| If <code>True</code>, an <code>Accelerator</code> or <code>PartialState</code> must be initialized. Note that by doing so, this could lead to issues | |
| with hyperparameter tuning.</li> | |
| </ul>`,name:"accelerator_config"},{anchor:"transformers.Seq2SeqTrainingArguments.label_smoothing_factor",description:`<strong>label_smoothing_factor</strong> (<code>float</code>, <em>optional</em>, defaults to 0.0) — | |
| The label smoothing factor to use. Zero means no label smoothing, otherwise the underlying onehot-encoded | |
| labels are changed from 0s and 1s to <code>label_smoothing_factor/num_labels</code> and <code>1 - label_smoothing_factor + label_smoothing_factor/num_labels</code> respectively.`,name:"label_smoothing_factor"},{anchor:"transformers.Seq2SeqTrainingArguments.debug",description:`<strong>debug</strong> (<code>str</code> or list of <code>DebugOption</code>, <em>optional</em>, defaults to <code>""</code>) — | |
| Enable one or more debug features. This is an experimental feature.</p> | |
| <p>Possible options are:</p> | |
| <ul> | |
| <li><code>"underflow_overflow"</code>: detects overflow in model’s input/outputs and reports the last frames that led to | |
| the event</li> | |
| <li><code>"tpu_metrics_debug"</code>: print debug metrics on TPU</li> | |
| </ul> | |
| <p>The options should be separated by whitespaces.`,name:"debug"},{anchor:"transformers.Seq2SeqTrainingArguments.optim",description:`<strong>optim</strong> (<code>str</code> or <code>training_args.OptimizerNames</code>, <em>optional</em>, defaults to <code>"adamw_torch"</code>) — | |
| The optimizer to use, such as “adamw_hf”, “adamw_torch”, “adamw_torch_fused”, “adamw_apex_fused”, “adamw_anyprecision”, | |
| “adafactor”. See <code>OptimizerNames</code> in <a href="https://github.com/huggingface/transformers/blob/main/src/transformers/training_args.py" rel="nofollow">training_args.py</a> | |
| for a full list of optimizers.`,name:"optim"},{anchor:"transformers.Seq2SeqTrainingArguments.optim_args",description:`<strong>optim_args</strong> (<code>str</code>, <em>optional</em>) — | |
| Optional arguments that are supplied to optimizers such as AnyPrecisionAdamW, AdEMAMix, and GaLore.`,name:"optim_args"},{anchor:"transformers.Seq2SeqTrainingArguments.group_by_length",description:`<strong>group_by_length</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to group together samples of roughly the same length in the training dataset (to minimize | |
| padding applied and be more efficient). Only useful if applying dynamic padding.`,name:"group_by_length"},{anchor:"transformers.Seq2SeqTrainingArguments.length_column_name",description:`<strong>length_column_name</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"length"</code>) — | |
| Column name for precomputed lengths. If the column exists, grouping by length will use these values rather | |
| than computing them on train startup. Ignored unless <code>group_by_length</code> is <code>True</code> and the dataset is an | |
| instance of <code>Dataset</code>.`,name:"length_column_name"},{anchor:"transformers.Seq2SeqTrainingArguments.report_to",description:`<strong>report_to</strong> (<code>str</code> or <code>List[str]</code>, <em>optional</em>, defaults to <code>"all"</code>) — | |
| The list of integrations to report the results and logs to. Supported platforms are <code>"azure_ml"</code>, | |
| <code>"clearml"</code>, <code>"codecarbon"</code>, <code>"comet_ml"</code>, <code>"dagshub"</code>, <code>"dvclive"</code>, <code>"flyte"</code>, <code>"mlflow"</code>, <code>"neptune"</code>, | |
| <code>"tensorboard"</code>, and <code>"wandb"</code>. Use <code>"all"</code> to report to all integrations installed, <code>"none"</code> for no | |
| integrations.`,name:"report_to"},{anchor:"transformers.Seq2SeqTrainingArguments.ddp_find_unused_parameters",description:`<strong>ddp_find_unused_parameters</strong> (<code>bool</code>, <em>optional</em>) — | |
| When using distributed training, the value of the flag <code>find_unused_parameters</code> passed to | |
| <code>DistributedDataParallel</code>. Will default to <code>False</code> if gradient checkpointing is used, <code>True</code> otherwise.`,name:"ddp_find_unused_parameters"},{anchor:"transformers.Seq2SeqTrainingArguments.ddp_bucket_cap_mb",description:`<strong>ddp_bucket_cap_mb</strong> (<code>int</code>, <em>optional</em>) — | |
| When using distributed training, the value of the flag <code>bucket_cap_mb</code> passed to <code>DistributedDataParallel</code>.`,name:"ddp_bucket_cap_mb"},{anchor:"transformers.Seq2SeqTrainingArguments.ddp_broadcast_buffers",description:`<strong>ddp_broadcast_buffers</strong> (<code>bool</code>, <em>optional</em>) — | |
| When using distributed training, the value of the flag <code>broadcast_buffers</code> passed to | |
| <code>DistributedDataParallel</code>. Will default to <code>False</code> if gradient checkpointing is used, <code>True</code> otherwise.`,name:"ddp_broadcast_buffers"},{anchor:"transformers.Seq2SeqTrainingArguments.dataloader_pin_memory",description:`<strong>dataloader_pin_memory</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether you want to pin memory in data loaders or not. Will default to <code>True</code>.`,name:"dataloader_pin_memory"},{anchor:"transformers.Seq2SeqTrainingArguments.dataloader_persistent_workers",description:`<strong>dataloader_persistent_workers</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| If True, the data loader will not shut down the worker processes after a dataset has been consumed once. | |
| This allows to maintain the workers Dataset instances alive. Can potentially speed up training, but will | |
| increase RAM usage. Will default to <code>False</code>.`,name:"dataloader_persistent_workers"},{anchor:"transformers.Seq2SeqTrainingArguments.dataloader_prefetch_factor",description:`<strong>dataloader_prefetch_factor</strong> (<code>int</code>, <em>optional</em>) — | |
| Number of batches loaded in advance by each worker. | |
| 2 means there will be a total of 2 * num_workers batches prefetched across all workers.`,name:"dataloader_prefetch_factor"},{anchor:"transformers.Seq2SeqTrainingArguments.skip_memory_metrics",description:`<strong>skip_memory_metrics</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to skip adding of memory profiler reports to metrics. This is skipped by default because it slows | |
| down the training and evaluation speed.`,name:"skip_memory_metrics"},{anchor:"transformers.Seq2SeqTrainingArguments.push_to_hub",description:`<strong>push_to_hub</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to push the model to the Hub every time the model is saved. If this is activated, | |
| <code>output_dir</code> will begin a git directory synced with the repo (determined by <code>hub_model_id</code>) and the content | |
| will be pushed each time a save is triggered (depending on your <code>save_strategy</code>). Calling | |
| <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer.save_model">save_model()</a> will also trigger a push.</p> | |
| <div class="course-tip course-tip-orange bg-gradient-to-br dark:bg-gradient-to-r before:border-orange-500 dark:before:border-orange-800 from-orange-50 dark:from-gray-900 to-white dark:to-gray-950 border border-orange-50 text-orange-700 dark:text-gray-400"> | |
| <p>If <code>output_dir</code> exists, it needs to be a local clone of the repository to which the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a> will be | |
| pushed.</p> | |
| </div>`,name:"push_to_hub"},{anchor:"transformers.Seq2SeqTrainingArguments.resume_from_checkpoint",description:`<strong>resume_from_checkpoint</strong> (<code>str</code>, <em>optional</em>) — | |
| The path to a folder with a valid checkpoint for your model. This argument is not directly used by | |
| <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>, it’s intended to be used by your training/evaluation scripts instead. See the <a href="https://github.com/huggingface/transformers/tree/main/examples" rel="nofollow">example | |
| scripts</a> for more details.`,name:"resume_from_checkpoint"},{anchor:"transformers.Seq2SeqTrainingArguments.hub_model_id",description:`<strong>hub_model_id</strong> (<code>str</code>, <em>optional</em>) — | |
| The name of the repository to keep in sync with the local <em>output_dir</em>. It can be a simple model ID in | |
| which case the model will be pushed in your namespace. Otherwise it should be the whole repository name, | |
| for instance <code>"user_name/model"</code>, which allows you to push to an organization you are a member of with | |
| <code>"organization_name/model"</code>. Will default to <code>user_name/output_dir_name</code> with <em>output_dir_name</em> being the | |
| name of <code>output_dir</code>.</p> | |
| <p>Will default to the name of <code>output_dir</code>.`,name:"hub_model_id"},{anchor:"transformers.Seq2SeqTrainingArguments.hub_strategy",description:`<strong>hub_strategy</strong> (<code>str</code> or <code>HubStrategy</code>, <em>optional</em>, defaults to <code>"every_save"</code>) — | |
| Defines the scope of what is pushed to the Hub and when. Possible values are:</p> | |
| <ul> | |
| <li><code>"end"</code>: push the model, its configuration, the processing class e.g. tokenizer (if passed along to the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>) and a | |
| draft of a model card when the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer.save_model">save_model()</a> method is called.</li> | |
| <li><code>"every_save"</code>: push the model, its configuration, the processing class e.g. tokenizer (if passed along to the <a href="/docs/transformers/pr_34009/ko/main_classes/trainer#transformers.Trainer">Trainer</a>) and | |
| a draft of a model card each time there is a model save. The pushes are asynchronous to not block | |
| training, and in case the save are very frequent, a new push is only attempted if the previous one is | |
| finished. A last push is made with the final model at the end of training.</li> | |
| <li><code>"checkpoint"</code>: like <code>"every_save"</code> but the latest checkpoint is also pushed in a subfolder named | |
| last-checkpoint, allowing you to resume training easily with | |
| <code>trainer.train(resume_from_checkpoint="last-checkpoint")</code>.</li> | |
| <li><code>"all_checkpoints"</code>: like <code>"checkpoint"</code> but all checkpoints are pushed like they appear in the output | |
| folder (so you will get one checkpoint folder per folder in your final repository)</li> | |
| </ul>`,name:"hub_strategy"},{anchor:"transformers.Seq2SeqTrainingArguments.hub_token",description:`<strong>hub_token</strong> (<code>str</code>, <em>optional</em>) — | |
| The token to use to push the model to the Hub. Will default to the token in the cache folder obtained with | |
| <code>huggingface-cli login</code>.`,name:"hub_token"},{anchor:"transformers.Seq2SeqTrainingArguments.hub_private_repo",description:`<strong>hub_private_repo</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| If True, the Hub repo will be set to private.`,name:"hub_private_repo"},{anchor:"transformers.Seq2SeqTrainingArguments.hub_always_push",description:`<strong>hub_always_push</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Unless this is <code>True</code>, the <code>Trainer</code> will skip pushing a checkpoint when the previous push is not finished.`,name:"hub_always_push"},{anchor:"transformers.Seq2SeqTrainingArguments.gradient_checkpointing",description:`<strong>gradient_checkpointing</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| If True, use gradient checkpointing to save memory at the expense of slower backward pass.`,name:"gradient_checkpointing"},{anchor:"transformers.Seq2SeqTrainingArguments.gradient_checkpointing_kwargs",description:`<strong>gradient_checkpointing_kwargs</strong> (<code>dict</code>, <em>optional</em>, defaults to <code>None</code>) — | |
| Key word arguments to be passed to the <code>gradient_checkpointing_enable</code> method.`,name:"gradient_checkpointing_kwargs"},{anchor:"transformers.Seq2SeqTrainingArguments.include_inputs_for_metrics",description:`<strong>include_inputs_for_metrics</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| This argument is deprecated. Use <code>include_for_metrics</code> instead, e.g, <code>include_for_metrics = ["inputs"]</code>.`,name:"include_inputs_for_metrics"},{anchor:"transformers.Seq2SeqTrainingArguments.include_for_metrics",description:`<strong>include_for_metrics</strong> (<code>List[str]</code>, <em>optional</em>, defaults to <code>[]</code>) — | |
| Include additional data in the <code>compute_metrics</code> function if needed for metrics computation. | |
| Possible options to add to <code>include_for_metrics</code> list:</p> | |
| <ul> | |
| <li><code>"inputs"</code>: Input data passed to the model, intended for calculating input dependent metrics.</li> | |
| <li><code>"loss"</code>: Loss values computed during evaluation, intended for calculating loss dependent metrics.</li> | |
| </ul>`,name:"include_for_metrics"},{anchor:"transformers.Seq2SeqTrainingArguments.eval_do_concat_batches",description:`<strong>eval_do_concat_batches</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>True</code>) — | |
| Whether to recursively concat inputs/losses/labels/predictions across batches. If <code>False</code>, | |
| will instead store them as lists, with each batch kept separate.`,name:"eval_do_concat_batches"},{anchor:"transformers.Seq2SeqTrainingArguments.auto_find_batch_size",description:`<strong>auto_find_batch_size</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to find a batch size that will fit into memory automatically through exponential decay, avoiding | |
| CUDA Out-of-Memory errors. Requires accelerate to be installed (<code>pip install accelerate</code>)`,name:"auto_find_batch_size"},{anchor:"transformers.Seq2SeqTrainingArguments.full_determinism",description:`<strong>full_determinism</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| If <code>True</code>, <code>enable_full_determinism()</code> is called instead of <code>set_seed()</code> to ensure reproducible results in | |
| distributed training. Important: this will negatively impact the performance, so only use it for debugging.`,name:"full_determinism"},{anchor:"transformers.Seq2SeqTrainingArguments.torchdynamo",description:`<strong>torchdynamo</strong> (<code>str</code>, <em>optional</em>) — | |
| If set, the backend compiler for TorchDynamo. Possible choices are <code>"eager"</code>, <code>"aot_eager"</code>, <code>"inductor"</code>, | |
| <code>"nvfuser"</code>, <code>"aot_nvfuser"</code>, <code>"aot_cudagraphs"</code>, <code>"ofi"</code>, <code>"fx2trt"</code>, <code>"onnxrt"</code> and <code>"ipex"</code>.`,name:"torchdynamo"},{anchor:"transformers.Seq2SeqTrainingArguments.ray_scope",description:`<strong>ray_scope</strong> (<code>str</code>, <em>optional</em>, defaults to <code>"last"</code>) — | |
| The scope to use when doing hyperparameter search with Ray. By default, <code>"last"</code> will be used. Ray will | |
| then use the last checkpoint of all trials, compare those, and select the best one. However, other options | |
| are also available. See the <a href="https://docs.ray.io/en/latest/tune/api_docs/analysis.html#ray.tune.ExperimentAnalysis.get_best_trial" rel="nofollow">Ray documentation</a> for | |
| more options.`,name:"ray_scope"},{anchor:"transformers.Seq2SeqTrainingArguments.ddp_timeout",description:`<strong>ddp_timeout</strong> (<code>int</code>, <em>optional</em>, defaults to 1800) — | |
| The timeout for <code>torch.distributed.init_process_group</code> calls, used to avoid GPU socket timeouts when | |
| performing slow operations in distributed runnings. Please refer the [PyTorch documentation] | |
| (<a href="https://pytorch.org/docs/stable/distributed.html#torch.distributed.init_process_group" rel="nofollow">https://pytorch.org/docs/stable/distributed.html#torch.distributed.init_process_group</a>) for more | |
| information.`,name:"ddp_timeout"},{anchor:"transformers.Seq2SeqTrainingArguments.use_mps_device",description:`<strong>use_mps_device</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| This argument is deprecated.<code>mps</code> device will be used if it is available similar to <code>cuda</code> device.`,name:"use_mps_device"},{anchor:"transformers.Seq2SeqTrainingArguments.torch_compile",description:`<strong>torch_compile</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether or not to compile the model using PyTorch 2.0 | |
| <a href="https://pytorch.org/get-started/pytorch-2.0/" rel="nofollow"><code>torch.compile</code></a>.</p> | |
| <p>This will use the best defaults for the <a href="https://pytorch.org/docs/stable/generated/torch.compile.html?highlight=torch+compile#torch.compile" rel="nofollow"><code>torch.compile</code> | |
| API</a>. | |
| You can customize the defaults with the argument <code>torch_compile_backend</code> and <code>torch_compile_mode</code> but we | |
| don’t guarantee any of them will work as the support is progressively rolled in in PyTorch.</p> | |
| <p>This flag and the whole compile API is experimental and subject to change in future releases.`,name:"torch_compile"},{anchor:"transformers.Seq2SeqTrainingArguments.torch_compile_backend",description:`<strong>torch_compile_backend</strong> (<code>str</code>, <em>optional</em>) — | |
| The backend to use in <code>torch.compile</code>. If set to any value, <code>torch_compile</code> will be set to <code>True</code>.</p> | |
| <p>Refer to the PyTorch doc for possible values and note that they may change across PyTorch versions.</p> | |
| <p>This flag is experimental and subject to change in future releases.`,name:"torch_compile_backend"},{anchor:"transformers.Seq2SeqTrainingArguments.torch_compile_mode",description:`<strong>torch_compile_mode</strong> (<code>str</code>, <em>optional</em>) — | |
| The mode to use in <code>torch.compile</code>. If set to any value, <code>torch_compile</code> will be set to <code>True</code>.</p> | |
| <p>Refer to the PyTorch doc for possible values and note that they may change across PyTorch versions.</p> | |
| <p>This flag is experimental and subject to change in future releases.`,name:"torch_compile_mode"},{anchor:"transformers.Seq2SeqTrainingArguments.split_batches",description:`<strong>split_batches</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not the accelerator should split the batches yielded by the dataloaders across the devices | |
| during distributed training. If</p> | |
| <p>set to <code>True</code>, the actual batch size used will be the same on any kind of distributed processes, but it | |
| must be a</p> | |
| <p>round multiple of the number of processes you are using (such as GPUs).`,name:"split_batches"},{anchor:"transformers.Seq2SeqTrainingArguments.include_tokens_per_second",description:`<strong>include_tokens_per_second</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to compute the number of tokens per second per device for training speed metrics.</p> | |
| <p>This will iterate over the entire training dataloader once beforehand,</p> | |
| <p>and will slow down the entire process.`,name:"include_tokens_per_second"},{anchor:"transformers.Seq2SeqTrainingArguments.include_num_input_tokens_seen",description:`<strong>include_num_input_tokens_seen</strong> (<code>bool</code>, <em>optional</em>) — | |
| Whether or not to track the number of input tokens seen throughout training.</p> | |
| <p>May be slower in distributed training as gather operations must be called.`,name:"include_num_input_tokens_seen"},{anchor:"transformers.Seq2SeqTrainingArguments.neftune_noise_alpha",description:`<strong>neftune_noise_alpha</strong> (<code>Optional[float]</code>) — | |
| If not <code>None</code>, this will activate NEFTune noise embeddings. This can drastically improve model performance | |
| for instruction fine-tuning. Check out the <a href="https://arxiv.org/abs/2310.05914" rel="nofollow">original paper</a> and the | |
| <a href="https://github.com/neelsjain/NEFTune" rel="nofollow">original code</a>. Support transformers <code>PreTrainedModel</code> and also | |
| <code>PeftModel</code> from peft. The original paper used values in the range [5.0, 15.0].`,name:"neftune_noise_alpha"},{anchor:"transformers.Seq2SeqTrainingArguments.optim_target_modules",description:`<strong>optim_target_modules</strong> (<code>Union[str, List[str]]</code>, <em>optional</em>) — | |
| The target modules to optimize, i.e. the module names that you would like to train, right now this is used only for GaLore algorithm | |
| <a href="https://arxiv.org/abs/2403.03507" rel="nofollow">https://arxiv.org/abs/2403.03507</a> | |
| See: <a href="https://github.com/jiaweizzhao/GaLore" rel="nofollow">https://github.com/jiaweizzhao/GaLore</a> for more details. You need to make sure to pass a valid GaloRe | |
| optimizer, e.g. one of: “galore_adamw”, “galore_adamw_8bit”, “galore_adafactor” and make sure that the target modules are <code>nn.Linear</code> modules | |
| only.`,name:"optim_target_modules"},{anchor:"transformers.Seq2SeqTrainingArguments.batch_eval_metrics",description:`<strong>batch_eval_metrics</strong> (<code>Optional[bool]</code>, defaults to <code>False</code>) — | |
| If set to <code>True</code>, evaluation will call compute_metrics at the end of each batch to accumulate statistics | |
| rather than saving all eval logits in memory. When set to <code>True</code>, you must pass a compute_metrics function | |
| that takes a boolean argument <code>compute_result</code>, which when passed <code>True</code>, will trigger the final global | |
| summary statistics from the batch-level summary statistics you’ve accumulated over the evaluation set.`,name:"batch_eval_metrics"},{anchor:"transformers.Seq2SeqTrainingArguments.eval_on_start",description:`<strong>eval_on_start</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to perform a evaluation step (sanity check) before the training to ensure the validation steps works correctly.`,name:"eval_on_start"},{anchor:"transformers.Seq2SeqTrainingArguments.eval_use_gather_object",description:`<strong>eval_use_gather_object</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to run recursively gather object in a nested list/tuple/dictionary of objects from all devices. This should only be enabled if users are not just returning tensors, and this is actively discouraged by PyTorch.`,name:"eval_use_gather_object"},{anchor:"transformers.Seq2SeqTrainingArguments.use_liger_kernel",description:`<strong>use_liger_kernel</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether enable <a href="https://github.com/linkedin/Liger-Kernel" rel="nofollow">Liger</a> Kernel for LLM model training. | |
| It can effectively increase multi-GPU training throughput by ~20% and reduces memory usage by ~60%, works out of the box with | |
| flash attention, PyTorch FSDP, and Microsoft DeepSpeed. Currently, it supports llama, mistral, mixtral and gemma models.`,name:"use_liger_kernel"},{anchor:"transformers.Seq2SeqTrainingArguments.sortish_sampler",description:`<strong>sortish_sampler</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to use a <em>sortish sampler</em> or not. Only possible if the underlying datasets are <em>Seq2SeqDataset</em> | |
| for now but will become generally available in the near future.</p> | |
| <p>It sorts the inputs according to lengths in order to minimize the padding size, with a bit of randomness | |
| for the training set.`,name:"sortish_sampler"},{anchor:"transformers.Seq2SeqTrainingArguments.predict_with_generate",description:`<strong>predict_with_generate</strong> (<code>bool</code>, <em>optional</em>, defaults to <code>False</code>) — | |
| Whether to use generate to calculate generative metrics (ROUGE, BLEU).`,name:"predict_with_generate"},{anchor:"transformers.Seq2SeqTrainingArguments.generation_max_length",description:`<strong>generation_max_length</strong> (<code>int</code>, <em>optional</em>) — | |
| The <code>max_length</code> to use on each evaluation loop when <code>predict_with_generate=True</code>. Will default to the | |
| <code>max_length</code> value of the model configuration.`,name:"generation_max_length"},{anchor:"transformers.Seq2SeqTrainingArguments.generation_num_beams",description:`<strong>generation_num_beams</strong> (<code>int</code>, <em>optional</em>) — | |
| The <code>num_beams</code> to use on each evaluation loop when <code>predict_with_generate=True</code>. Will default to the | |
| <code>num_beams</code> value of the model configuration.`,name:"generation_num_beams"},{anchor:"transformers.Seq2SeqTrainingArguments.generation_config",description:`<strong>generation_config</strong> (<code>str</code> or <code>Path</code> or <code>GenerationConfig</code>, <em>optional</em>) — | |
| Allows to load a <code>GenerationConfig</code> from the <code>from_pretrained</code> method. This can be either:</p> | |
| <ul> | |
| <li>a string, the <em>model id</em> of a pretrained model configuration hosted inside a model repo on | |
| huggingface.co.</li> | |
| <li>a path to a <em>directory</em> containing a configuration file saved using the | |
| <code>save_pretrained()</code> method, e.g., <code>./my_model_directory/</code>.</li> | |
| <li>a <code>GenerationConfig</code> object.</li> | |
| </ul>`,name:"generation_config"}],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args_seq2seq.py#L28"}}),Do=new x({props:{name:"to_dict",anchor:"transformers.Seq2SeqTrainingArguments.to_dict",parameters:[],source:"https://github.com/huggingface/transformers/blob/vr_34009/src/transformers/training_args_seq2seq.py#L86"}}),Io=new ym({props:{source:"https://github.com/huggingface/transformers/blob/main/docs/source/ko/main_classes/trainer.md"}}),{c(){i=r("meta"),k=o(),m=r("p"),c=o(),p(q.$$.fragment),s=o(),A=r("p"),A.innerHTML=md,ga=o(),Tt=r("p"),Tt.innerHTML=pd,ha=o(),p($e.$$.fragment),fa=o(),p(wt.$$.fragment),_a=o(),b=r("div"),p(xt.$$.fragment),Ra=o(),Bo=r("p"),Bo.textContent=ud,Ja=o(),Go=r("p"),Go.textContent=gd,Ea=o(),Vo=r("ul"),Vo.innerHTML=hd,Ba=o(),Ae=r("div"),p(qt.$$.fragment),Ga=o(),Xo=r("p"),Xo.innerHTML=fd,Va=o(),Se=r("div"),p(kt.$$.fragment),Xa=o(),Zo=r("p"),Zo.innerHTML=_d,Za=o(),X=r("div"),p($t.$$.fragment),Ya=o(),Yo=r("p"),Yo.textContent=bd,Qa=o(),Qo=r("p"),Qo.textContent=vd,Ka=o(),Ce=r("div"),p(At.$$.fragment),es=o(),Ko=r("p"),Ko.textContent=yd,ts=o(),Pe=r("div"),p(St.$$.fragment),os=o(),en=r("p"),en.innerHTML=Td,ns=o(),Z=r("div"),p(Ct.$$.fragment),rs=o(),tn=r("p"),tn.textContent=wd,as=o(),on=r("p"),on.innerHTML=xd,ss=o(),Y=r("div"),p(Pt.$$.fragment),is=o(),nn=r("p"),nn.textContent=qd,ls=o(),rn=r("p"),rn.innerHTML=kd,ds=o(),Me=r("div"),p(Mt.$$.fragment),cs=o(),an=r("p"),an.textContent=$d,ms=o(),L=r("div"),p(Ft.$$.fragment),ps=o(),sn=r("p"),sn.textContent=Ad,us=o(),ln=r("p"),ln.innerHTML=Sd,gs=o(),dn=r("p"),dn.textContent=Cd,hs=o(),Q=r("div"),p(zt.$$.fragment),fs=o(),cn=r("p"),cn.innerHTML=Pd,_s=o(),mn=r("p"),mn.textContent=Md,bs=o(),Fe=r("div"),p(Dt.$$.fragment),vs=o(),pn=r("p"),pn.innerHTML=Fd,ys=o(),K=r("div"),p(It.$$.fragment),Ts=o(),un=r("p"),un.textContent=zd,ws=o(),gn=r("p"),gn.textContent=Dd,xs=o(),ee=r("div"),p(Ut.$$.fragment),qs=o(),hn=r("p"),hn.innerHTML=Id,ks=o(),fn=r("p"),fn.textContent=Ud,$s=o(),ze=r("div"),p(Wt.$$.fragment),As=o(),_n=r("p"),_n.textContent=Wd,Ss=o(),De=r("div"),p(Lt.$$.fragment),Cs=o(),bn=r("p"),bn.textContent=Ld,Ps=o(),Ie=r("div"),p(Nt.$$.fragment),Ms=o(),vn=r("p"),vn.textContent=Nd,Fs=o(),Ue=r("div"),p(jt.$$.fragment),zs=o(),yn=r("p"),yn.textContent=jd,Ds=o(),te=r("div"),p(Ot.$$.fragment),Is=o(),Tn=r("p"),Tn.innerHTML=Od,Us=o(),wn=r("p"),wn.textContent=Hd,Ws=o(),N=r("div"),p(Ht.$$.fragment),Ls=o(),xn=r("p"),xn.innerHTML=Rd,Ns=o(),qn=r("p"),qn.innerHTML=Jd,js=o(),kn=r("p"),kn.textContent=Ed,Os=o(),oe=r("div"),p(Rt.$$.fragment),Hs=o(),$n=r("p"),$n.innerHTML=Bd,Rs=o(),p(We.$$.fragment),Js=o(),Le=r("div"),p(Jt.$$.fragment),Es=o(),An=r("p"),An.innerHTML=Gd,Bs=o(),Ne=r("div"),p(Et.$$.fragment),Gs=o(),Sn=r("p"),Sn.textContent=Vd,Vs=o(),je=r("div"),p(Bt.$$.fragment),Xs=o(),Cn=r("p"),Cn.innerHTML=Xd,Zs=o(),ne=r("div"),p(Gt.$$.fragment),Ys=o(),Pn=r("p"),Pn.innerHTML=Zd,Qs=o(),Mn=r("p"),Mn.textContent=Yd,Ks=o(),P=r("div"),p(Vt.$$.fragment),ei=o(),Fn=r("p"),Fn.textContent=Qd,ti=o(),zn=r("p"),zn.textContent=Kd,oi=o(),Dn=r("p"),Dn.textContent=ec,ni=o(),In=r("p"),In.innerHTML=tc,ri=o(),p(Oe.$$.fragment),ai=o(),Un=r("p"),Un.innerHTML=oc,si=o(),Wn=r("ul"),Wn.innerHTML=nc,ii=o(),Ln=r("p"),Ln.textContent=rc,li=o(),Nn=r("p"),Nn.textContent=ac,di=o(),jn=r("p"),jn.innerHTML=sc,ci=o(),On=r("p"),On.innerHTML=ic,mi=o(),Hn=r("p"),Hn.innerHTML=lc,pi=o(),Rn=r("p"),Rn.innerHTML=dc,ui=o(),Jn=r("p"),Jn.innerHTML=cc,gi=o(),En=r("p"),En.textContent=mc,hi=o(),He=r("div"),p(Xt.$$.fragment),fi=o(),Bn=r("p"),Bn.textContent=pc,_i=o(),Re=r("div"),p(Zt.$$.fragment),bi=o(),Gn=r("p"),Gn.innerHTML=uc,vi=o(),Je=r("div"),p(Yt.$$.fragment),yi=o(),Vn=r("p"),Vn.innerHTML=gc,Ti=o(),re=r("div"),p(Qt.$$.fragment),wi=o(),Xn=r("p"),Xn.innerHTML=hc,xi=o(),Zn=r("p"),Zn.innerHTML=fc,qi=o(),D=r("div"),p(Kt.$$.fragment),ki=o(),Yn=r("p"),Yn.textContent=_c,$i=o(),Qn=r("p"),Qn.innerHTML=bc,Ai=o(),p(Ee.$$.fragment),Si=o(),Kn=r("p"),Kn.innerHTML=vc,Ci=o(),er=r("ul"),er.innerHTML=yc,Pi=o(),ae=r("div"),p(eo.$$.fragment),Mi=o(),tr=r("p"),tr.innerHTML=Tc,Fi=o(),or=r("p"),or.textContent=wc,zi=o(),se=r("div"),p(to.$$.fragment),Di=o(),nr=r("p"),nr.innerHTML=xc,Ii=o(),rr=r("p"),rr.textContent=qc,Ui=o(),Be=r("div"),p(oo.$$.fragment),Wi=o(),ar=r("p"),ar.textContent=kc,Li=o(),Ge=r("div"),p(no.$$.fragment),Ni=o(),sr=r("p"),sr.innerHTML=$c,ji=o(),Ve=r("div"),p(ro.$$.fragment),Oi=o(),ir=r("p"),ir.innerHTML=Ac,Hi=o(),j=r("div"),p(ao.$$.fragment),Ri=o(),lr=r("p"),lr.innerHTML=Sc,Ji=o(),dr=r("p"),dr.textContent=Cc,Ei=o(),cr=r("p"),cr.innerHTML=Pc,Bi=o(),ie=r("div"),p(so.$$.fragment),Gi=o(),mr=r("p"),mr.innerHTML=Mc,Vi=o(),pr=r("p"),pr.textContent=Fc,Xi=o(),le=r("div"),p(io.$$.fragment),Zi=o(),ur=r("p"),ur.textContent=zc,Yi=o(),gr=r("p"),gr.textContent=Dc,Qi=o(),Xe=r("div"),p(lo.$$.fragment),Ki=o(),hr=r("p"),hr.textContent=Ic,el=o(),de=r("div"),p(co.$$.fragment),tl=o(),fr=r("p"),fr.textContent=Uc,ol=o(),_r=r("p"),_r.textContent=Wc,ba=o(),p(mo.$$.fragment),va=o(),G=r("div"),p(po.$$.fragment),nl=o(),O=r("div"),p(uo.$$.fragment),rl=o(),br=r("p"),br.textContent=Lc,al=o(),vr=r("p"),vr.innerHTML=Nc,sl=o(),yr=r("p"),yr.textContent=jc,il=o(),I=r("div"),p(go.$$.fragment),ll=o(),Tr=r("p"),Tr.textContent=Oc,dl=o(),wr=r("p"),wr.innerHTML=Hc,cl=o(),p(Ze.$$.fragment),ml=o(),xr=r("p"),xr.innerHTML=Rc,pl=o(),qr=r("ul"),qr.innerHTML=Jc,ya=o(),p(ho.$$.fragment),Ta=o(),S=r("div"),p(fo.$$.fragment),ul=o(),kr=r("p"),kr.innerHTML=Ec,gl=o(),$r=r("p"),$r.innerHTML=Bc,hl=o(),U=r("div"),p(_o.$$.fragment),fl=o(),Ar=r("p"),Ar.textContent=Gc,_l=o(),Sr=r("p"),Sr.innerHTML=Vc,bl=o(),Cr=r("p"),Cr.innerHTML=Xc,vl=o(),Pr=r("p"),Pr.innerHTML=Zc,yl=o(),Ye=r("div"),p(bo.$$.fragment),Tl=o(),Mr=r("p"),Mr.textContent=Yc,wl=o(),ce=r("div"),p(vo.$$.fragment),xl=o(),Fr=r("p"),Fr.textContent=Qc,ql=o(),zr=r("p"),zr.innerHTML=Kc,kl=o(),me=r("div"),p(yo.$$.fragment),$l=o(),Dr=r("p"),Dr.textContent=em,Al=o(),p(Qe.$$.fragment),Sl=o(),pe=r("div"),p(To.$$.fragment),Cl=o(),Ir=r("p"),Ir.textContent=tm,Pl=o(),p(Ke.$$.fragment),Ml=o(),ue=r("div"),p(wo.$$.fragment),Fl=o(),Ur=r("p"),Ur.textContent=om,zl=o(),p(et.$$.fragment),Dl=o(),ge=r("div"),p(xo.$$.fragment),Il=o(),Wr=r("p"),Wr.textContent=nm,Ul=o(),p(tt.$$.fragment),Wl=o(),he=r("div"),p(qo.$$.fragment),Ll=o(),Lr=r("p"),Lr.textContent=rm,Nl=o(),p(ot.$$.fragment),jl=o(),H=r("div"),p(ko.$$.fragment),Ol=o(),Nr=r("p"),Nr.textContent=am,Hl=o(),p(nt.$$.fragment),Rl=o(),p(rt.$$.fragment),Jl=o(),fe=r("div"),p($o.$$.fragment),El=o(),jr=r("p"),jr.textContent=sm,Bl=o(),p(at.$$.fragment),Gl=o(),R=r("div"),p(Ao.$$.fragment),Vl=o(),Or=r("p"),Or.textContent=im,Xl=o(),p(st.$$.fragment),Zl=o(),p(it.$$.fragment),Yl=o(),J=r("div"),p(So.$$.fragment),Ql=o(),Hr=r("p"),Hr.textContent=lm,Kl=o(),p(lt.$$.fragment),ed=o(),p(dt.$$.fragment),td=o(),ct=r("div"),p(Co.$$.fragment),od=o(),Rr=r("p"),Rr.innerHTML=dm,nd=o(),mt=r("div"),p(Po.$$.fragment),rd=o(),Jr=r("p"),Jr.textContent=cm,ad=o(),pt=r("div"),p(Mo.$$.fragment),sd=o(),Er=r("p"),Er.textContent=mm,wa=o(),p(Fo.$$.fragment),xa=o(),W=r("div"),p(zo.$$.fragment),id=o(),Br=r("p"),Br.innerHTML=pm,ld=o(),Gr=r("p"),Gr.innerHTML=um,dd=o(),ut=r("div"),p(Do.$$.fragment),cd=o(),Vr=r("p"),Vr.innerHTML=gm,qa=o(),p(Io.$$.fragment),ka=o(),pa=r("p"),this.h()},l(t){const y=vm("svelte-u9bgzb",document.head);i=a(y,"META",{name:!0,content:!0}),y.forEach(d),k=n(t),m=a(t,"P",{}),T(m).forEach(d),c=n(t),u(q.$$.fragment,t),s=n(t),A=a(t,"P",{"data-svelte-h":!0}),l(A)!=="svelte-wzlr5w"&&(A.innerHTML=md),ga=n(t),Tt=a(t,"P",{"data-svelte-h":!0}),l(Tt)!=="svelte-jp82nf"&&(Tt.innerHTML=pd),ha=n(t),u($e.$$.fragment,t),fa=n(t),u(wt.$$.fragment,t),_a=n(t),b=a(t,"DIV",{class:!0});var v=T(b);u(xt.$$.fragment,v),Ra=n(v),Bo=a(v,"P",{"data-svelte-h":!0}),l(Bo)!=="svelte-xa06uv"&&(Bo.textContent=ud),Ja=n(v),Go=a(v,"P",{"data-svelte-h":!0}),l(Go)!=="svelte-1tbd587"&&(Go.textContent=gd),Ea=n(v),Vo=a(v,"UL",{"data-svelte-h":!0}),l(Vo)!=="svelte-1szbyww"&&(Vo.innerHTML=hd),Ba=n(v),Ae=a(v,"DIV",{class:!0});var Uo=T(Ae);u(qt.$$.fragment,Uo),Ga=n(Uo),Xo=a(Uo,"P",{"data-svelte-h":!0}),l(Xo)!=="svelte-erlepi"&&(Xo.innerHTML=fd),Uo.forEach(d),Va=n(v),Se=a(v,"DIV",{class:!0});var Wo=T(Se);u(kt.$$.fragment,Wo),Xa=n(Wo),Zo=a(Wo,"P",{"data-svelte-h":!0}),l(Zo)!=="svelte-9kwitf"&&(Zo.innerHTML=_d),Wo.forEach(d),Za=n(v),X=a(v,"DIV",{class:!0});var ye=T(X);u($t.$$.fragment,ye),Ya=n(ye),Yo=a(ye,"P",{"data-svelte-h":!0}),l(Yo)!=="svelte-1feheyo"&&(Yo.textContent=bd),Qa=n(ye),Qo=a(ye,"P",{"data-svelte-h":!0}),l(Qo)!=="svelte-1be1oij"&&(Qo.textContent=vd),ye.forEach(d),Ka=n(v),Ce=a(v,"DIV",{class:!0});var Lo=T(Ce);u(At.$$.fragment,Lo),es=n(Lo),Ko=a(Lo,"P",{"data-svelte-h":!0}),l(Ko)!=="svelte-1agg8ch"&&(Ko.textContent=yd),Lo.forEach(d),ts=n(v),Pe=a(v,"DIV",{class:!0});var No=T(Pe);u(St.$$.fragment,No),os=n(No),en=a(No,"P",{"data-svelte-h":!0}),l(en)!=="svelte-1mh859w"&&(en.innerHTML=Td),No.forEach(d),ns=n(v),Z=a(v,"DIV",{class:!0});var Te=T(Z);u(Ct.$$.fragment,Te),rs=n(Te),tn=a(Te,"P",{"data-svelte-h":!0}),l(tn)!=="svelte-1w438iv"&&(tn.textContent=wd),as=n(Te),on=a(Te,"P",{"data-svelte-h":!0}),l(on)!=="svelte-nadplw"&&(on.innerHTML=xd),Te.forEach(d),ss=n(v),Y=a(v,"DIV",{class:!0});var we=T(Y);u(Pt.$$.fragment,we),is=n(we),nn=a(we,"P",{"data-svelte-h":!0}),l(nn)!=="svelte-wdco14"&&(nn.textContent=qd),ls=n(we),rn=a(we,"P",{"data-svelte-h":!0}),l(rn)!=="svelte-zasfqr"&&(rn.innerHTML=kd),we.forEach(d),ds=n(v),Me=a(v,"DIV",{class:!0});var jo=T(Me);u(Mt.$$.fragment,jo),cs=n(jo),an=a(jo,"P",{"data-svelte-h":!0}),l(an)!=="svelte-qi4mtl"&&(an.textContent=$d),jo.forEach(d),ms=n(v),L=a(v,"DIV",{class:!0});var V=T(L);u(Ft.$$.fragment,V),ps=n(V),sn=a(V,"P",{"data-svelte-h":!0}),l(sn)!=="svelte-1mcvryw"&&(sn.textContent=Ad),us=n(V),ln=a(V,"P",{"data-svelte-h":!0}),l(ln)!=="svelte-1339kj6"&&(ln.innerHTML=Sd),gs=n(V),dn=a(V,"P",{"data-svelte-h":!0}),l(dn)!=="svelte-1h30111"&&(dn.textContent=Cd),V.forEach(d),hs=n(v),Q=a(v,"DIV",{class:!0});var xe=T(Q);u(zt.$$.fragment,xe),fs=n(xe),cn=a(xe,"P",{"data-svelte-h":!0}),l(cn)!=="svelte-1v5gmc1"&&(cn.innerHTML=Pd),_s=n(xe),mn=a(xe,"P",{"data-svelte-h":!0}),l(mn)!=="svelte-1tyo99t"&&(mn.textContent=Md),xe.forEach(d),bs=n(v),Fe=a(v,"DIV",{class:!0});var Oo=T(Fe);u(Dt.$$.fragment,Oo),vs=n(Oo),pn=a(Oo,"P",{"data-svelte-h":!0}),l(pn)!=="svelte-1f8grfw"&&(pn.innerHTML=Fd),Oo.forEach(d),ys=n(v),K=a(v,"DIV",{class:!0});var qe=T(K);u(It.$$.fragment,qe),Ts=n(qe),un=a(qe,"P",{"data-svelte-h":!0}),l(un)!=="svelte-yguuq6"&&(un.textContent=zd),ws=n(qe),gn=a(qe,"P",{"data-svelte-h":!0}),l(gn)!=="svelte-452p7a"&&(gn.textContent=Dd),qe.forEach(d),xs=n(v),ee=a(v,"DIV",{class:!0});var ke=T(ee);u(Ut.$$.fragment,ke),qs=n(ke),hn=a(ke,"P",{"data-svelte-h":!0}),l(hn)!=="svelte-xesobz"&&(hn.innerHTML=Id),ks=n(ke),fn=a(ke,"P",{"data-svelte-h":!0}),l(fn)!=="svelte-1v6sccr"&&(fn.textContent=Ud),ke.forEach(d),$s=n(v),ze=a(v,"DIV",{class:!0});var Ho=T(ze);u(Wt.$$.fragment,Ho),As=n(Ho),_n=a(Ho,"P",{"data-svelte-h":!0}),l(_n)!=="svelte-fo0w3k"&&(_n.textContent=Wd),Ho.forEach(d),Ss=n(v),De=a(v,"DIV",{class:!0});var Ro=T(De);u(Lt.$$.fragment,Ro),Cs=n(Ro),bn=a(Ro,"P",{"data-svelte-h":!0}),l(bn)!=="svelte-n48d75"&&(bn.textContent=Ld),Ro.forEach(d),Ps=n(v),Ie=a(v,"DIV",{class:!0});var Jo=T(Ie);u(Nt.$$.fragment,Jo),Ms=n(Jo),vn=a(Jo,"P",{"data-svelte-h":!0}),l(vn)!=="svelte-1enq7q"&&(vn.textContent=Nd),Jo.forEach(d),Fs=n(v),Ue=a(v,"DIV",{class:!0});var Aa=T(Ue);u(jt.$$.fragment,Aa),zs=n(Aa),yn=a(Aa,"P",{"data-svelte-h":!0}),l(yn)!=="svelte-4yzrvz"&&(yn.textContent=jd),Aa.forEach(d),Ds=n(v),te=a(v,"DIV",{class:!0});var Xr=T(te);u(Ot.$$.fragment,Xr),Is=n(Xr),Tn=a(Xr,"P",{"data-svelte-h":!0}),l(Tn)!=="svelte-upvtbd"&&(Tn.innerHTML=Od),Us=n(Xr),wn=a(Xr,"P",{"data-svelte-h":!0}),l(wn)!=="svelte-1v6sccr"&&(wn.textContent=Hd),Xr.forEach(d),Ws=n(v),N=a(v,"DIV",{class:!0});var gt=T(N);u(Ht.$$.fragment,gt),Ls=n(gt),xn=a(gt,"P",{"data-svelte-h":!0}),l(xn)!=="svelte-dkae9b"&&(xn.innerHTML=Rd),Ns=n(gt),qn=a(gt,"P",{"data-svelte-h":!0}),l(qn)!=="svelte-2uvyde"&&(qn.innerHTML=Jd),js=n(gt),kn=a(gt,"P",{"data-svelte-h":!0}),l(kn)!=="svelte-1v6sccr"&&(kn.textContent=Ed),gt.forEach(d),Os=n(v),oe=a(v,"DIV",{class:!0});var Zr=T(oe);u(Rt.$$.fragment,Zr),Hs=n(Zr),$n=a(Zr,"P",{"data-svelte-h":!0}),l($n)!=="svelte-mf62ka"&&($n.innerHTML=Bd),Rs=n(Zr),u(We.$$.fragment,Zr),Zr.forEach(d),Js=n(v),Le=a(v,"DIV",{class:!0});var Sa=T(Le);u(Jt.$$.fragment,Sa),Es=n(Sa),An=a(Sa,"P",{"data-svelte-h":!0}),l(An)!=="svelte-rjlxwe"&&(An.innerHTML=Gd),Sa.forEach(d),Bs=n(v),Ne=a(v,"DIV",{class:!0});var Ca=T(Ne);u(Et.$$.fragment,Ca),Gs=n(Ca),Sn=a(Ca,"P",{"data-svelte-h":!0}),l(Sn)!=="svelte-16imdke"&&(Sn.textContent=Vd),Ca.forEach(d),Vs=n(v),je=a(v,"DIV",{class:!0});var Pa=T(je);u(Bt.$$.fragment,Pa),Xs=n(Pa),Cn=a(Pa,"P",{"data-svelte-h":!0}),l(Cn)!=="svelte-1s7jdq8"&&(Cn.innerHTML=Xd),Pa.forEach(d),Zs=n(v),ne=a(v,"DIV",{class:!0});var Yr=T(ne);u(Gt.$$.fragment,Yr),Ys=n(Yr),Pn=a(Yr,"P",{"data-svelte-h":!0}),l(Pn)!=="svelte-19fk55y"&&(Pn.innerHTML=Zd),Qs=n(Yr),Mn=a(Yr,"P",{"data-svelte-h":!0}),l(Mn)!=="svelte-1tz9xhn"&&(Mn.textContent=Yd),Yr.forEach(d),Ks=n(v),P=a(v,"DIV",{class:!0});var F=T(P);u(Vt.$$.fragment,F),ei=n(F),Fn=a(F,"P",{"data-svelte-h":!0}),l(Fn)!=="svelte-1q4m3ey"&&(Fn.textContent=Qd),ti=n(F),zn=a(F,"P",{"data-svelte-h":!0}),l(zn)!=="svelte-5vrny9"&&(zn.textContent=Kd),oi=n(F),Dn=a(F,"P",{"data-svelte-h":!0}),l(Dn)!=="svelte-k63f70"&&(Dn.textContent=ec),ni=n(F),In=a(F,"P",{"data-svelte-h":!0}),l(In)!=="svelte-y2g1lw"&&(In.innerHTML=tc),ri=n(F),u(Oe.$$.fragment,F),ai=n(F),Un=a(F,"P",{"data-svelte-h":!0}),l(Un)!=="svelte-ny4g3p"&&(Un.innerHTML=oc),si=n(F),Wn=a(F,"UL",{"data-svelte-h":!0}),l(Wn)!=="svelte-9luch7"&&(Wn.innerHTML=nc),ii=n(F),Ln=a(F,"P",{"data-svelte-h":!0}),l(Ln)!=="svelte-100tlvw"&&(Ln.textContent=rc),li=n(F),Nn=a(F,"P",{"data-svelte-h":!0}),l(Nn)!=="svelte-1d82c6u"&&(Nn.textContent=ac),di=n(F),jn=a(F,"P",{"data-svelte-h":!0}),l(jn)!=="svelte-d1rr2j"&&(jn.innerHTML=sc),ci=n(F),On=a(F,"P",{"data-svelte-h":!0}),l(On)!=="svelte-m7v5wt"&&(On.innerHTML=ic),mi=n(F),Hn=a(F,"P",{"data-svelte-h":!0}),l(Hn)!=="svelte-12aktao"&&(Hn.innerHTML=lc),pi=n(F),Rn=a(F,"P",{"data-svelte-h":!0}),l(Rn)!=="svelte-1lmh4zl"&&(Rn.innerHTML=dc),ui=n(F),Jn=a(F,"P",{"data-svelte-h":!0}),l(Jn)!=="svelte-1gqy7ut"&&(Jn.innerHTML=cc),gi=n(F),En=a(F,"P",{"data-svelte-h":!0}),l(En)!=="svelte-1ocu6t3"&&(En.textContent=mc),F.forEach(d),hi=n(v),He=a(v,"DIV",{class:!0});var Ma=T(He);u(Xt.$$.fragment,Ma),fi=n(Ma),Bn=a(Ma,"P",{"data-svelte-h":!0}),l(Bn)!=="svelte-otcx5b"&&(Bn.textContent=pc),Ma.forEach(d),_i=n(v),Re=a(v,"DIV",{class:!0});var Fa=T(Re);u(Zt.$$.fragment,Fa),bi=n(Fa),Gn=a(Fa,"P",{"data-svelte-h":!0}),l(Gn)!=="svelte-19inmpu"&&(Gn.innerHTML=uc),Fa.forEach(d),vi=n(v),Je=a(v,"DIV",{class:!0});var za=T(Je);u(Yt.$$.fragment,za),yi=n(za),Vn=a(za,"P",{"data-svelte-h":!0}),l(Vn)!=="svelte-tkw811"&&(Vn.innerHTML=gc),za.forEach(d),Ti=n(v),re=a(v,"DIV",{class:!0});var Qr=T(re);u(Qt.$$.fragment,Qr),wi=n(Qr),Xn=a(Qr,"P",{"data-svelte-h":!0}),l(Xn)!=="svelte-11dvmkl"&&(Xn.innerHTML=hc),xi=n(Qr),Zn=a(Qr,"P",{"data-svelte-h":!0}),l(Zn)!=="svelte-1ojzwr1"&&(Zn.innerHTML=fc),Qr.forEach(d),qi=n(v),D=a(v,"DIV",{class:!0});var E=T(D);u(Kt.$$.fragment,E),ki=n(E),Yn=a(E,"P",{"data-svelte-h":!0}),l(Yn)!=="svelte-1b6cpoy"&&(Yn.textContent=_c),$i=n(E),Qn=a(E,"P",{"data-svelte-h":!0}),l(Qn)!=="svelte-o30dtk"&&(Qn.innerHTML=bc),Ai=n(E),u(Ee.$$.fragment,E),Si=n(E),Kn=a(E,"P",{"data-svelte-h":!0}),l(Kn)!=="svelte-1t1ea6f"&&(Kn.innerHTML=vc),Ci=n(E),er=a(E,"UL",{"data-svelte-h":!0}),l(er)!=="svelte-15ctb4a"&&(er.innerHTML=yc),E.forEach(d),Pi=n(v),ae=a(v,"DIV",{class:!0});var Kr=T(ae);u(eo.$$.fragment,Kr),Mi=n(Kr),tr=a(Kr,"P",{"data-svelte-h":!0}),l(tr)!=="svelte-1v5gmc1"&&(tr.innerHTML=Tc),Fi=n(Kr),or=a(Kr,"P",{"data-svelte-h":!0}),l(or)!=="svelte-1tyo99t"&&(or.textContent=wc),Kr.forEach(d),zi=n(v),se=a(v,"DIV",{class:!0});var ea=T(se);u(to.$$.fragment,ea),Di=n(ea),nr=a(ea,"P",{"data-svelte-h":!0}),l(nr)!=="svelte-1y56pc5"&&(nr.innerHTML=xc),Ii=n(ea),rr=a(ea,"P",{"data-svelte-h":!0}),l(rr)!=="svelte-qjfexy"&&(rr.textContent=qc),ea.forEach(d),Ui=n(v),Be=a(v,"DIV",{class:!0});var Da=T(Be);u(oo.$$.fragment,Da),Wi=n(Da),ar=a(Da,"P",{"data-svelte-h":!0}),l(ar)!=="svelte-o5qc7e"&&(ar.textContent=kc),Da.forEach(d),Li=n(v),Ge=a(v,"DIV",{class:!0});var Ia=T(Ge);u(no.$$.fragment,Ia),Ni=n(Ia),sr=a(Ia,"P",{"data-svelte-h":!0}),l(sr)!=="svelte-8tudwd"&&(sr.innerHTML=$c),Ia.forEach(d),ji=n(v),Ve=a(v,"DIV",{class:!0});var Ua=T(Ve);u(ro.$$.fragment,Ua),Oi=n(Ua),ir=a(Ua,"P",{"data-svelte-h":!0}),l(ir)!=="svelte-1hz0n8s"&&(ir.innerHTML=Ac),Ua.forEach(d),Hi=n(v),j=a(v,"DIV",{class:!0});var ht=T(j);u(ao.$$.fragment,ht),Ri=n(ht),lr=a(ht,"P",{"data-svelte-h":!0}),l(lr)!=="svelte-9dohab"&&(lr.innerHTML=Sc),Ji=n(ht),dr=a(ht,"P",{"data-svelte-h":!0}),l(dr)!=="svelte-5vrny9"&&(dr.textContent=Cc),Ei=n(ht),cr=a(ht,"P",{"data-svelte-h":!0}),l(cr)!=="svelte-172s2u0"&&(cr.innerHTML=Pc),ht.forEach(d),Bi=n(v),ie=a(v,"DIV",{class:!0});var ta=T(ie);u(so.$$.fragment,ta),Gi=n(ta),mr=a(ta,"P",{"data-svelte-h":!0}),l(mr)!=="svelte-r8h4ov"&&(mr.innerHTML=Mc),Vi=n(ta),pr=a(ta,"P",{"data-svelte-h":!0}),l(pr)!=="svelte-1e6bius"&&(pr.textContent=Fc),ta.forEach(d),Xi=n(v),le=a(v,"DIV",{class:!0});var oa=T(le);u(io.$$.fragment,oa),Zi=n(oa),ur=a(oa,"P",{"data-svelte-h":!0}),l(ur)!=="svelte-1g3hgx9"&&(ur.textContent=zc),Yi=n(oa),gr=a(oa,"P",{"data-svelte-h":!0}),l(gr)!=="svelte-5vrny9"&&(gr.textContent=Dc),oa.forEach(d),Qi=n(v),Xe=a(v,"DIV",{class:!0});var Wa=T(Xe);u(lo.$$.fragment,Wa),Ki=n(Wa),hr=a(Wa,"P",{"data-svelte-h":!0}),l(hr)!=="svelte-1cilnet"&&(hr.textContent=Ic),Wa.forEach(d),el=n(v),de=a(v,"DIV",{class:!0});var na=T(de);u(co.$$.fragment,na),tl=n(na),fr=a(na,"P",{"data-svelte-h":!0}),l(fr)!=="svelte-4qplkw"&&(fr.textContent=Uc),ol=n(na),_r=a(na,"P",{"data-svelte-h":!0}),l(_r)!=="svelte-qjfexy"&&(_r.textContent=Wc),na.forEach(d),v.forEach(d),ba=n(t),u(mo.$$.fragment,t),va=n(t),G=a(t,"DIV",{class:!0});var ra=T(G);u(po.$$.fragment,ra),nl=n(ra),O=a(ra,"DIV",{class:!0});var ft=T(O);u(uo.$$.fragment,ft),rl=n(ft),br=a(ft,"P",{"data-svelte-h":!0}),l(br)!=="svelte-1mcvryw"&&(br.textContent=Lc),al=n(ft),vr=a(ft,"P",{"data-svelte-h":!0}),l(vr)!=="svelte-1339kj6"&&(vr.innerHTML=Nc),sl=n(ft),yr=a(ft,"P",{"data-svelte-h":!0}),l(yr)!=="svelte-1h30111"&&(yr.textContent=jc),ft.forEach(d),il=n(ra),I=a(ra,"DIV",{class:!0});var B=T(I);u(go.$$.fragment,B),ll=n(B),Tr=a(B,"P",{"data-svelte-h":!0}),l(Tr)!=="svelte-1b6cpoy"&&(Tr.textContent=Oc),dl=n(B),wr=a(B,"P",{"data-svelte-h":!0}),l(wr)!=="svelte-o30dtk"&&(wr.innerHTML=Hc),cl=n(B),u(Ze.$$.fragment,B),ml=n(B),xr=a(B,"P",{"data-svelte-h":!0}),l(xr)!=="svelte-1t1ea6f"&&(xr.innerHTML=Rc),pl=n(B),qr=a(B,"UL",{"data-svelte-h":!0}),l(qr)!=="svelte-15ctb4a"&&(qr.innerHTML=Jc),B.forEach(d),ra.forEach(d),ya=n(t),u(ho.$$.fragment,t),Ta=n(t),S=a(t,"DIV",{class:!0});var M=T(S);u(fo.$$.fragment,M),ul=n(M),kr=a(M,"P",{"data-svelte-h":!0}),l(kr)!=="svelte-avfxjl"&&(kr.innerHTML=Ec),gl=n(M),$r=a(M,"P",{"data-svelte-h":!0}),l($r)!=="svelte-1xl7jqc"&&($r.innerHTML=Bc),hl=n(M),U=a(M,"DIV",{class:!0});var _e=T(U);u(_o.$$.fragment,_e),fl=n(_e),Ar=a(_e,"P",{"data-svelte-h":!0}),l(Ar)!=="svelte-hj8f41"&&(Ar.textContent=Gc),_l=n(_e),Sr=a(_e,"P",{"data-svelte-h":!0}),l(Sr)!=="svelte-1iq1lj0"&&(Sr.innerHTML=Vc),bl=n(_e),Cr=a(_e,"P",{"data-svelte-h":!0}),l(Cr)!=="svelte-18bal86"&&(Cr.innerHTML=Xc),vl=n(_e),Pr=a(_e,"P",{"data-svelte-h":!0}),l(Pr)!=="svelte-oacj0p"&&(Pr.innerHTML=Zc),_e.forEach(d),yl=n(M),Ye=a(M,"DIV",{class:!0});var La=T(Ye);u(bo.$$.fragment,La),Tl=n(La),Mr=a(La,"P",{"data-svelte-h":!0}),l(Mr)!=="svelte-9pbc1"&&(Mr.textContent=Yc),La.forEach(d),wl=n(M),ce=a(M,"DIV",{class:!0});var aa=T(ce);u(vo.$$.fragment,aa),xl=n(aa),Fr=a(aa,"P",{"data-svelte-h":!0}),l(Fr)!=="svelte-13fy4lg"&&(Fr.textContent=Qc),ql=n(aa),zr=a(aa,"P",{"data-svelte-h":!0}),l(zr)!=="svelte-1aa165w"&&(zr.innerHTML=Kc),aa.forEach(d),kl=n(M),me=a(M,"DIV",{class:!0});var sa=T(me);u(yo.$$.fragment,sa),$l=n(sa),Dr=a(sa,"P",{"data-svelte-h":!0}),l(Dr)!=="svelte-12p68tr"&&(Dr.textContent=em),Al=n(sa),u(Qe.$$.fragment,sa),sa.forEach(d),Sl=n(M),pe=a(M,"DIV",{class:!0});var ia=T(pe);u(To.$$.fragment,ia),Cl=n(ia),Ir=a(ia,"P",{"data-svelte-h":!0}),l(Ir)!=="svelte-9ywrxl"&&(Ir.textContent=tm),Pl=n(ia),u(Ke.$$.fragment,ia),ia.forEach(d),Ml=n(M),ue=a(M,"DIV",{class:!0});var la=T(ue);u(wo.$$.fragment,la),Fl=n(la),Ur=a(la,"P",{"data-svelte-h":!0}),l(Ur)!=="svelte-1lwqsi8"&&(Ur.textContent=om),zl=n(la),u(et.$$.fragment,la),la.forEach(d),Dl=n(M),ge=a(M,"DIV",{class:!0});var da=T(ge);u(xo.$$.fragment,da),Il=n(da),Wr=a(da,"P",{"data-svelte-h":!0}),l(Wr)!=="svelte-mc54aq"&&(Wr.textContent=nm),Ul=n(da),u(tt.$$.fragment,da),da.forEach(d),Wl=n(M),he=a(M,"DIV",{class:!0});var ca=T(he);u(qo.$$.fragment,ca),Ll=n(ca),Lr=a(ca,"P",{"data-svelte-h":!0}),l(Lr)!=="svelte-18hco78"&&(Lr.textContent=rm),Nl=n(ca),u(ot.$$.fragment,ca),ca.forEach(d),jl=n(M),H=a(M,"DIV",{class:!0});var _t=T(H);u(ko.$$.fragment,_t),Ol=n(_t),Nr=a(_t,"P",{"data-svelte-h":!0}),l(Nr)!=="svelte-y1tcex"&&(Nr.textContent=am),Hl=n(_t),u(nt.$$.fragment,_t),Rl=n(_t),u(rt.$$.fragment,_t),_t.forEach(d),Jl=n(M),fe=a(M,"DIV",{class:!0});var ma=T(fe);u($o.$$.fragment,ma),El=n(ma),jr=a(ma,"P",{"data-svelte-h":!0}),l(jr)!=="svelte-1duj623"&&(jr.textContent=sm),Bl=n(ma),u(at.$$.fragment,ma),ma.forEach(d),Gl=n(M),R=a(M,"DIV",{class:!0});var bt=T(R);u(Ao.$$.fragment,bt),Vl=n(bt),Or=a(bt,"P",{"data-svelte-h":!0}),l(Or)!=="svelte-lhuo6p"&&(Or.textContent=im),Xl=n(bt),u(st.$$.fragment,bt),Zl=n(bt),u(it.$$.fragment,bt),bt.forEach(d),Yl=n(M),J=a(M,"DIV",{class:!0});var vt=T(J);u(So.$$.fragment,vt),Ql=n(vt),Hr=a(vt,"P",{"data-svelte-h":!0}),l(Hr)!=="svelte-1h65v06"&&(Hr.textContent=lm),Kl=n(vt),u(lt.$$.fragment,vt),ed=n(vt),u(dt.$$.fragment,vt),vt.forEach(d),td=n(M),ct=a(M,"DIV",{class:!0});var Na=T(ct);u(Co.$$.fragment,Na),od=n(Na),Rr=a(Na,"P",{"data-svelte-h":!0}),l(Rr)!=="svelte-1ciob6u"&&(Rr.innerHTML=dm),Na.forEach(d),nd=n(M),mt=a(M,"DIV",{class:!0});var ja=T(mt);u(Po.$$.fragment,ja),rd=n(ja),Jr=a(ja,"P",{"data-svelte-h":!0}),l(Jr)!=="svelte-5ayq1f"&&(Jr.textContent=cm),ja.forEach(d),ad=n(M),pt=a(M,"DIV",{class:!0});var Oa=T(pt);u(Mo.$$.fragment,Oa),sd=n(Oa),Er=a(Oa,"P",{"data-svelte-h":!0}),l(Er)!=="svelte-lg45ly"&&(Er.textContent=mm),Oa.forEach(d),M.forEach(d),wa=n(t),u(Fo.$$.fragment,t),xa=n(t),W=a(t,"DIV",{class:!0});var yt=T(W);u(zo.$$.fragment,yt),id=n(yt),Br=a(yt,"P",{"data-svelte-h":!0}),l(Br)!=="svelte-avfxjl"&&(Br.innerHTML=pm),ld=n(yt),Gr=a(yt,"P",{"data-svelte-h":!0}),l(Gr)!=="svelte-1xl7jqc"&&(Gr.innerHTML=um),dd=n(yt),ut=a(yt,"DIV",{class:!0});var Ha=T(ut);u(Do.$$.fragment,Ha),cd=n(Ha),Vr=a(Ha,"P",{"data-svelte-h":!0}),l(Vr)!=="svelte-yrflrv"&&(Vr.innerHTML=gm),Ha.forEach(d),yt.forEach(d),qa=n(t),u(Io.$$.fragment,t),ka=n(t),pa=a(t,"P",{}),T(pa).forEach(d),this.h()},h(){w(i,"name","hf:doc:metadata"),w(i,"content",Nm),w(Ae,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Se,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(X,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Ce,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Pe,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Z,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Y,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Me,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(L,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Q,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Fe,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(K,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(ee,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(ze,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(De,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Ie,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Ue,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(te,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(N,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(oe,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Le,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Ne,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(je,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(ne,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(P,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(He,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Re,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Je,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(re,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(D,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(ae,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(se,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Be,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Ge,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Ve,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(j,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(ie,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(le,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Xe,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(de,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(b,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(O,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(I,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(G,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(U,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(Ye,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(ce,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(me,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(pe,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(ue,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(ge,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(he,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(H,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(fe,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(R,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(J,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(ct,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(mt,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(pt,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(S,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(ut,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"),w(W,"class","docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8")},m(t,y){e(document.head,i),$(t,k,y),$(t,m,y),$(t,c,y),g(q,t,y),$(t,s,y),$(t,A,y),$(t,ga,y),$(t,Tt,y),$(t,ha,y),g($e,t,y),$(t,fa,y),g(wt,t,y),$(t,_a,y),$(t,b,y),g(xt,b,null),e(b,Ra),e(b,Bo),e(b,Ja),e(b,Go),e(b,Ea),e(b,Vo),e(b,Ba),e(b,Ae),g(qt,Ae,null),e(Ae,Ga),e(Ae,Xo),e(b,Va),e(b,Se),g(kt,Se,null),e(Se,Xa),e(Se,Zo),e(b,Za),e(b,X),g($t,X,null),e(X,Ya),e(X,Yo),e(X,Qa),e(X,Qo),e(b,Ka),e(b,Ce),g(At,Ce,null),e(Ce,es),e(Ce,Ko),e(b,ts),e(b,Pe),g(St,Pe,null),e(Pe,os),e(Pe,en),e(b,ns),e(b,Z),g(Ct,Z,null),e(Z,rs),e(Z,tn),e(Z,as),e(Z,on),e(b,ss),e(b,Y),g(Pt,Y,null),e(Y,is),e(Y,nn),e(Y,ls),e(Y,rn),e(b,ds),e(b,Me),g(Mt,Me,null),e(Me,cs),e(Me,an),e(b,ms),e(b,L),g(Ft,L,null),e(L,ps),e(L,sn),e(L,us),e(L,ln),e(L,gs),e(L,dn),e(b,hs),e(b,Q),g(zt,Q,null),e(Q,fs),e(Q,cn),e(Q,_s),e(Q,mn),e(b,bs),e(b,Fe),g(Dt,Fe,null),e(Fe,vs),e(Fe,pn),e(b,ys),e(b,K),g(It,K,null),e(K,Ts),e(K,un),e(K,ws),e(K,gn),e(b,xs),e(b,ee),g(Ut,ee,null),e(ee,qs),e(ee,hn),e(ee,ks),e(ee,fn),e(b,$s),e(b,ze),g(Wt,ze,null),e(ze,As),e(ze,_n),e(b,Ss),e(b,De),g(Lt,De,null),e(De,Cs),e(De,bn),e(b,Ps),e(b,Ie),g(Nt,Ie,null),e(Ie,Ms),e(Ie,vn),e(b,Fs),e(b,Ue),g(jt,Ue,null),e(Ue,zs),e(Ue,yn),e(b,Ds),e(b,te),g(Ot,te,null),e(te,Is),e(te,Tn),e(te,Us),e(te,wn),e(b,Ws),e(b,N),g(Ht,N,null),e(N,Ls),e(N,xn),e(N,Ns),e(N,qn),e(N,js),e(N,kn),e(b,Os),e(b,oe),g(Rt,oe,null),e(oe,Hs),e(oe,$n),e(oe,Rs),g(We,oe,null),e(b,Js),e(b,Le),g(Jt,Le,null),e(Le,Es),e(Le,An),e(b,Bs),e(b,Ne),g(Et,Ne,null),e(Ne,Gs),e(Ne,Sn),e(b,Vs),e(b,je),g(Bt,je,null),e(je,Xs),e(je,Cn),e(b,Zs),e(b,ne),g(Gt,ne,null),e(ne,Ys),e(ne,Pn),e(ne,Qs),e(ne,Mn),e(b,Ks),e(b,P),g(Vt,P,null),e(P,ei),e(P,Fn),e(P,ti),e(P,zn),e(P,oi),e(P,Dn),e(P,ni),e(P,In),e(P,ri),g(Oe,P,null),e(P,ai),e(P,Un),e(P,si),e(P,Wn),e(P,ii),e(P,Ln),e(P,li),e(P,Nn),e(P,di),e(P,jn),e(P,ci),e(P,On),e(P,mi),e(P,Hn),e(P,pi),e(P,Rn),e(P,ui),e(P,Jn),e(P,gi),e(P,En),e(b,hi),e(b,He),g(Xt,He,null),e(He,fi),e(He,Bn),e(b,_i),e(b,Re),g(Zt,Re,null),e(Re,bi),e(Re,Gn),e(b,vi),e(b,Je),g(Yt,Je,null),e(Je,yi),e(Je,Vn),e(b,Ti),e(b,re),g(Qt,re,null),e(re,wi),e(re,Xn),e(re,xi),e(re,Zn),e(b,qi),e(b,D),g(Kt,D,null),e(D,ki),e(D,Yn),e(D,$i),e(D,Qn),e(D,Ai),g(Ee,D,null),e(D,Si),e(D,Kn),e(D,Ci),e(D,er),e(b,Pi),e(b,ae),g(eo,ae,null),e(ae,Mi),e(ae,tr),e(ae,Fi),e(ae,or),e(b,zi),e(b,se),g(to,se,null),e(se,Di),e(se,nr),e(se,Ii),e(se,rr),e(b,Ui),e(b,Be),g(oo,Be,null),e(Be,Wi),e(Be,ar),e(b,Li),e(b,Ge),g(no,Ge,null),e(Ge,Ni),e(Ge,sr),e(b,ji),e(b,Ve),g(ro,Ve,null),e(Ve,Oi),e(Ve,ir),e(b,Hi),e(b,j),g(ao,j,null),e(j,Ri),e(j,lr),e(j,Ji),e(j,dr),e(j,Ei),e(j,cr),e(b,Bi),e(b,ie),g(so,ie,null),e(ie,Gi),e(ie,mr),e(ie,Vi),e(ie,pr),e(b,Xi),e(b,le),g(io,le,null),e(le,Zi),e(le,ur),e(le,Yi),e(le,gr),e(b,Qi),e(b,Xe),g(lo,Xe,null),e(Xe,Ki),e(Xe,hr),e(b,el),e(b,de),g(co,de,null),e(de,tl),e(de,fr),e(de,ol),e(de,_r),$(t,ba,y),g(mo,t,y),$(t,va,y),$(t,G,y),g(po,G,null),e(G,nl),e(G,O),g(uo,O,null),e(O,rl),e(O,br),e(O,al),e(O,vr),e(O,sl),e(O,yr),e(G,il),e(G,I),g(go,I,null),e(I,ll),e(I,Tr),e(I,dl),e(I,wr),e(I,cl),g(Ze,I,null),e(I,ml),e(I,xr),e(I,pl),e(I,qr),$(t,ya,y),g(ho,t,y),$(t,Ta,y),$(t,S,y),g(fo,S,null),e(S,ul),e(S,kr),e(S,gl),e(S,$r),e(S,hl),e(S,U),g(_o,U,null),e(U,fl),e(U,Ar),e(U,_l),e(U,Sr),e(U,bl),e(U,Cr),e(U,vl),e(U,Pr),e(S,yl),e(S,Ye),g(bo,Ye,null),e(Ye,Tl),e(Ye,Mr),e(S,wl),e(S,ce),g(vo,ce,null),e(ce,xl),e(ce,Fr),e(ce,ql),e(ce,zr),e(S,kl),e(S,me),g(yo,me,null),e(me,$l),e(me,Dr),e(me,Al),g(Qe,me,null),e(S,Sl),e(S,pe),g(To,pe,null),e(pe,Cl),e(pe,Ir),e(pe,Pl),g(Ke,pe,null),e(S,Ml),e(S,ue),g(wo,ue,null),e(ue,Fl),e(ue,Ur),e(ue,zl),g(et,ue,null),e(S,Dl),e(S,ge),g(xo,ge,null),e(ge,Il),e(ge,Wr),e(ge,Ul),g(tt,ge,null),e(S,Wl),e(S,he),g(qo,he,null),e(he,Ll),e(he,Lr),e(he,Nl),g(ot,he,null),e(S,jl),e(S,H),g(ko,H,null),e(H,Ol),e(H,Nr),e(H,Hl),g(nt,H,null),e(H,Rl),g(rt,H,null),e(S,Jl),e(S,fe),g($o,fe,null),e(fe,El),e(fe,jr),e(fe,Bl),g(at,fe,null),e(S,Gl),e(S,R),g(Ao,R,null),e(R,Vl),e(R,Or),e(R,Xl),g(st,R,null),e(R,Zl),g(it,R,null),e(S,Yl),e(S,J),g(So,J,null),e(J,Ql),e(J,Hr),e(J,Kl),g(lt,J,null),e(J,ed),g(dt,J,null),e(S,td),e(S,ct),g(Co,ct,null),e(ct,od),e(ct,Rr),e(S,nd),e(S,mt),g(Po,mt,null),e(mt,rd),e(mt,Jr),e(S,ad),e(S,pt),g(Mo,pt,null),e(pt,sd),e(pt,Er),$(t,wa,y),g(Fo,t,y),$(t,xa,y),$(t,W,y),g(zo,W,null),e(W,id),e(W,Br),e(W,ld),e(W,Gr),e(W,dd),e(W,ut),g(Do,ut,null),e(ut,cd),e(ut,Vr),$(t,qa,y),g(Io,t,y),$(t,ka,y),$(t,pa,y),$a=!0},p(t,[y]){const v={};y&2&&(v.$$scope={dirty:y,ctx:t}),$e.$set(v);const Uo={};y&2&&(Uo.$$scope={dirty:y,ctx:t}),We.$set(Uo);const Wo={};y&2&&(Wo.$$scope={dirty:y,ctx:t}),Oe.$set(Wo);const ye={};y&2&&(ye.$$scope={dirty:y,ctx:t}),Ee.$set(ye);const Lo={};y&2&&(Lo.$$scope={dirty:y,ctx:t}),Ze.$set(Lo);const No={};y&2&&(No.$$scope={dirty:y,ctx:t}),Qe.$set(No);const Te={};y&2&&(Te.$$scope={dirty:y,ctx:t}),Ke.$set(Te);const we={};y&2&&(we.$$scope={dirty:y,ctx:t}),et.$set(we);const jo={};y&2&&(jo.$$scope={dirty:y,ctx:t}),tt.$set(jo);const V={};y&2&&(V.$$scope={dirty:y,ctx:t}),ot.$set(V);const xe={};y&2&&(xe.$$scope={dirty:y,ctx:t}),nt.$set(xe);const Oo={};y&2&&(Oo.$$scope={dirty:y,ctx:t}),rt.$set(Oo);const qe={};y&2&&(qe.$$scope={dirty:y,ctx:t}),at.$set(qe);const ke={};y&2&&(ke.$$scope={dirty:y,ctx:t}),st.$set(ke);const Ho={};y&2&&(Ho.$$scope={dirty:y,ctx:t}),it.$set(Ho);const Ro={};y&2&&(Ro.$$scope={dirty:y,ctx:t}),lt.$set(Ro);const Jo={};y&2&&(Jo.$$scope={dirty:y,ctx:t}),dt.$set(Jo)},i(t){$a||(h(q.$$.fragment,t),h($e.$$.fragment,t),h(wt.$$.fragment,t),h(xt.$$.fragment,t),h(qt.$$.fragment,t),h(kt.$$.fragment,t),h($t.$$.fragment,t),h(At.$$.fragment,t),h(St.$$.fragment,t),h(Ct.$$.fragment,t),h(Pt.$$.fragment,t),h(Mt.$$.fragment,t),h(Ft.$$.fragment,t),h(zt.$$.fragment,t),h(Dt.$$.fragment,t),h(It.$$.fragment,t),h(Ut.$$.fragment,t),h(Wt.$$.fragment,t),h(Lt.$$.fragment,t),h(Nt.$$.fragment,t),h(jt.$$.fragment,t),h(Ot.$$.fragment,t),h(Ht.$$.fragment,t),h(Rt.$$.fragment,t),h(We.$$.fragment,t),h(Jt.$$.fragment,t),h(Et.$$.fragment,t),h(Bt.$$.fragment,t),h(Gt.$$.fragment,t),h(Vt.$$.fragment,t),h(Oe.$$.fragment,t),h(Xt.$$.fragment,t),h(Zt.$$.fragment,t),h(Yt.$$.fragment,t),h(Qt.$$.fragment,t),h(Kt.$$.fragment,t),h(Ee.$$.fragment,t),h(eo.$$.fragment,t),h(to.$$.fragment,t),h(oo.$$.fragment,t),h(no.$$.fragment,t),h(ro.$$.fragment,t),h(ao.$$.fragment,t),h(so.$$.fragment,t),h(io.$$.fragment,t),h(lo.$$.fragment,t),h(co.$$.fragment,t),h(mo.$$.fragment,t),h(po.$$.fragment,t),h(uo.$$.fragment,t),h(go.$$.fragment,t),h(Ze.$$.fragment,t),h(ho.$$.fragment,t),h(fo.$$.fragment,t),h(_o.$$.fragment,t),h(bo.$$.fragment,t),h(vo.$$.fragment,t),h(yo.$$.fragment,t),h(Qe.$$.fragment,t),h(To.$$.fragment,t),h(Ke.$$.fragment,t),h(wo.$$.fragment,t),h(et.$$.fragment,t),h(xo.$$.fragment,t),h(tt.$$.fragment,t),h(qo.$$.fragment,t),h(ot.$$.fragment,t),h(ko.$$.fragment,t),h(nt.$$.fragment,t),h(rt.$$.fragment,t),h($o.$$.fragment,t),h(at.$$.fragment,t),h(Ao.$$.fragment,t),h(st.$$.fragment,t),h(it.$$.fragment,t),h(So.$$.fragment,t),h(lt.$$.fragment,t),h(dt.$$.fragment,t),h(Co.$$.fragment,t),h(Po.$$.fragment,t),h(Mo.$$.fragment,t),h(Fo.$$.fragment,t),h(zo.$$.fragment,t),h(Do.$$.fragment,t),h(Io.$$.fragment,t),$a=!0)},o(t){f(q.$$.fragment,t),f($e.$$.fragment,t),f(wt.$$.fragment,t),f(xt.$$.fragment,t),f(qt.$$.fragment,t),f(kt.$$.fragment,t),f($t.$$.fragment,t),f(At.$$.fragment,t),f(St.$$.fragment,t),f(Ct.$$.fragment,t),f(Pt.$$.fragment,t),f(Mt.$$.fragment,t),f(Ft.$$.fragment,t),f(zt.$$.fragment,t),f(Dt.$$.fragment,t),f(It.$$.fragment,t),f(Ut.$$.fragment,t),f(Wt.$$.fragment,t),f(Lt.$$.fragment,t),f(Nt.$$.fragment,t),f(jt.$$.fragment,t),f(Ot.$$.fragment,t),f(Ht.$$.fragment,t),f(Rt.$$.fragment,t),f(We.$$.fragment,t),f(Jt.$$.fragment,t),f(Et.$$.fragment,t),f(Bt.$$.fragment,t),f(Gt.$$.fragment,t),f(Vt.$$.fragment,t),f(Oe.$$.fragment,t),f(Xt.$$.fragment,t),f(Zt.$$.fragment,t),f(Yt.$$.fragment,t),f(Qt.$$.fragment,t),f(Kt.$$.fragment,t),f(Ee.$$.fragment,t),f(eo.$$.fragment,t),f(to.$$.fragment,t),f(oo.$$.fragment,t),f(no.$$.fragment,t),f(ro.$$.fragment,t),f(ao.$$.fragment,t),f(so.$$.fragment,t),f(io.$$.fragment,t),f(lo.$$.fragment,t),f(co.$$.fragment,t),f(mo.$$.fragment,t),f(po.$$.fragment,t),f(uo.$$.fragment,t),f(go.$$.fragment,t),f(Ze.$$.fragment,t),f(ho.$$.fragment,t),f(fo.$$.fragment,t),f(_o.$$.fragment,t),f(bo.$$.fragment,t),f(vo.$$.fragment,t),f(yo.$$.fragment,t),f(Qe.$$.fragment,t),f(To.$$.fragment,t),f(Ke.$$.fragment,t),f(wo.$$.fragment,t),f(et.$$.fragment,t),f(xo.$$.fragment,t),f(tt.$$.fragment,t),f(qo.$$.fragment,t),f(ot.$$.fragment,t),f(ko.$$.fragment,t),f(nt.$$.fragment,t),f(rt.$$.fragment,t),f($o.$$.fragment,t),f(at.$$.fragment,t),f(Ao.$$.fragment,t),f(st.$$.fragment,t),f(it.$$.fragment,t),f(So.$$.fragment,t),f(lt.$$.fragment,t),f(dt.$$.fragment,t),f(Co.$$.fragment,t),f(Po.$$.fragment,t),f(Mo.$$.fragment,t),f(Fo.$$.fragment,t),f(zo.$$.fragment,t),f(Do.$$.fragment,t),f(Io.$$.fragment,t),$a=!1},d(t){t&&(d(k),d(m),d(c),d(s),d(A),d(ga),d(Tt),d(ha),d(fa),d(_a),d(b),d(ba),d(va),d(G),d(ya),d(Ta),d(S),d(wa),d(xa),d(W),d(qa),d(ka),d(pa)),d(i),_(q,t),_($e,t),_(wt,t),_(xt),_(qt),_(kt),_($t),_(At),_(St),_(Ct),_(Pt),_(Mt),_(Ft),_(zt),_(Dt),_(It),_(Ut),_(Wt),_(Lt),_(Nt),_(jt),_(Ot),_(Ht),_(Rt),_(We),_(Jt),_(Et),_(Bt),_(Gt),_(Vt),_(Oe),_(Xt),_(Zt),_(Yt),_(Qt),_(Kt),_(Ee),_(eo),_(to),_(oo),_(no),_(ro),_(ao),_(so),_(io),_(lo),_(co),_(mo,t),_(po),_(uo),_(go),_(Ze),_(ho,t),_(fo),_(_o),_(bo),_(vo),_(yo),_(Qe),_(To),_(Ke),_(wo),_(et),_(xo),_(tt),_(qo),_(ot),_(ko),_(nt),_(rt),_($o),_(at),_(Ao),_(st),_(it),_(So),_(lt),_(dt),_(Co),_(Po),_(Mo),_(Fo,t),_(zo),_(Do),_(Io,t)}}}const Nm='{"title":"Trainer","local":"trainer","sections":[{"title":"Trainer","local":"transformers.Trainer ][ transformers.Trainer","sections":[],"depth":2},{"title":"Seq2SeqTrainer","local":"transformers.Seq2SeqTrainer ][ transformers.Seq2SeqTrainer","sections":[],"depth":2},{"title":"TrainingArguments","local":"transformers.TrainingArguments ][ transformers.TrainingArguments","sections":[],"depth":2},{"title":"Seq2SeqTrainingArguments","local":"transformers.Seq2SeqTrainingArguments ][ transformers.Seq2SeqTrainingArguments","sections":[],"depth":2}],"depth":1}';function jm(C){return fm(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class Vm extends _m{constructor(i){super(),bm(this,i,jm,Lm,hm,{})}}export{Vm as component}; | |
Xet Storage Details
- Size:
- 295 kB
- Xet hash:
- 187af85cbe2c006a17fb834afce6a9be66cbf93a2cf8387e753ef3c8aca4782e
·
Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.