Buckets:

HuggingFaceDocBuilder's picture
download
raw
6.04 kB
import{s as tt,n as et,o as at}from"../chunks/scheduler.d75c11ed.js";import{S as st,i as ot,e as l,s as o,c as q,h as nt,a as i,d as a,b as n,f as Z,g as z,j as p,k as $,l as lt,m as s,n as F,t as N,o as R,p as U}from"../chunks/index.4ec9dfe9.js";import{C as it,H as rt,E as dt}from"../chunks/MermaidChart.svelte_svelte_type_style_lang.72f30ad6.js";function ct(G){let r,C,_,x,h,L,f,T,m,Y=`Each dataset should have a dataset card to promote responsible usage and inform users of any potential biases within the dataset.
This idea was inspired by the Model Cards proposed by <a href="https://huggingface.co/papers/1810.03993" rel="nofollow">Mitchell, 2018</a>.
Dataset cards help users understand a dataset’s contents, the context for using the dataset, how it was created, and any other considerations a user should be aware of.`,k,g,B="Creating a dataset card is easy and can be done in just a few steps:",H,y,K='<li><p>Go to your dataset repository on the <a href="https://hf.co/new-dataset" rel="nofollow">Hub</a> and click on <strong>Create Dataset Card</strong> to create a new <code>README.md</code> file in your repository.</p></li> <li><p>Use the <strong>Metadata UI</strong> to select the tags that describe your dataset. You can add a license, language, pretty_name, the task_categories, size_categories, and any other tags that you think are relevant. These tags help users discover and find your dataset on the Hub.</p></li>',E,d,Q='<img class="block dark:hidden" src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/hub/datasets-metadata-ui.png"/> <img class="hidden dark:block" src="https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/hub/datasets-metadata-ui-dark.png"/>',P,c,V='<p>For a complete, but not required, set of tag options you can also look at the <a href="https://github.com/huggingface/hub-docs/blob/main/datasetcard.md?plain=1" rel="nofollow">Dataset Card specifications</a>. This’ll have a few more tag options like <code>multilinguality</code> and <code>language_creators</code> which are useful but not absolutely necessary.</p>',D,u,J='<li><p>Click on the <strong>Import dataset card template</strong> link to automatically create a template with all the relevant fields to complete. Fill out the template sections to the best of your ability. Take a look at the <a href="https://github.com/huggingface/datasets/blob/main/templates/README_guide.md" rel="nofollow">Dataset Card Creation Guide</a> for more detailed information about what to include in each section of the card. For fields you are unable to complete, you can write <strong>[More Information Needed]</strong>.</p></li> <li><p>Once you’re done, commit the changes to the <code>README.md</code> file and you’ll see the completed dataset card on your repository.</p></li>',A,b,W='YAML also allows you to customize the way your dataset is loaded by <a href="./repository_structure#define-your-splits-and-subsets-in-yaml">defining splits and/or configurations</a> without the need to write any code.',O,v,X='Feel free to take a look at the <a href="https://huggingface.co/datasets/stanfordnlp/snli" rel="nofollow">SNLI</a>, <a href="https://huggingface.co/datasets/abisee/cnn_dailymail" rel="nofollow">CNN/DailyMail</a>, and <a href="https://huggingface.co/datasets/tblard/allocine" rel="nofollow">Allociné</a> dataset cards as examples to help you get started.',I,w,S,M,j;return h=new it({props:{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"}}),f=new rt({props:{title:"Create a dataset card",local:"create-a-dataset-card",headingTag:"h1"}}),w=new dt({props:{source:"https://github.com/huggingface/datasets/blob/main/docs/source/dataset_card.mdx"}}),{c(){r=l("meta"),C=o(),_=l("p"),x=o(),q(h.$$.fragment),L=o(),q(f.$$.fragment),T=o(),m=l("p"),m.innerHTML=Y,k=o(),g=l("p"),g.textContent=B,H=o(),y=l("ol"),y.innerHTML=K,E=o(),d=l("div"),d.innerHTML=Q,P=o(),c=l("blockquote"),c.innerHTML=V,D=o(),u=l("ol"),u.innerHTML=J,A=o(),b=l("p"),b.innerHTML=W,O=o(),v=l("p"),v.innerHTML=X,I=o(),q(w.$$.fragment),S=o(),M=l("p"),this.h()},l(t){const e=nt("svelte-u9bgzb",document.head);r=i(e,"META",{name:!0,content:!0}),e.forEach(a),C=n(t),_=i(t,"P",{}),Z(_).forEach(a),x=n(t),z(h.$$.fragment,t),L=n(t),z(f.$$.fragment,t),T=n(t),m=i(t,"P",{"data-svelte-h":!0}),p(m)!=="svelte-k2tnti"&&(m.innerHTML=Y),k=n(t),g=i(t,"P",{"data-svelte-h":!0}),p(g)!=="svelte-1ia96fv"&&(g.textContent=B),H=n(t),y=i(t,"OL",{"data-svelte-h":!0}),p(y)!=="svelte-9w9t8g"&&(y.innerHTML=K),E=n(t),d=i(t,"DIV",{class:!0,"data-svelte-h":!0}),p(d)!=="svelte-78myk1"&&(d.innerHTML=Q),P=n(t),c=i(t,"BLOCKQUOTE",{class:!0,"data-svelte-h":!0}),p(c)!=="svelte-i7cnv"&&(c.innerHTML=V),D=n(t),u=i(t,"OL",{start:!0,"data-svelte-h":!0}),p(u)!=="svelte-tamvby"&&(u.innerHTML=J),A=n(t),b=i(t,"P",{"data-svelte-h":!0}),p(b)!=="svelte-ttvehl"&&(b.innerHTML=W),O=n(t),v=i(t,"P",{"data-svelte-h":!0}),p(v)!=="svelte-aljo7a"&&(v.innerHTML=X),I=n(t),z(w.$$.fragment,t),S=n(t),M=i(t,"P",{}),Z(M).forEach(a),this.h()},h(){$(r,"name","hf:doc:metadata"),$(r,"content",ut),$(d,"class","flex justify-center"),$(c,"class","tip"),$(u,"start","3")},m(t,e){lt(document.head,r),s(t,C,e),s(t,_,e),s(t,x,e),F(h,t,e),s(t,L,e),F(f,t,e),s(t,T,e),s(t,m,e),s(t,k,e),s(t,g,e),s(t,H,e),s(t,y,e),s(t,E,e),s(t,d,e),s(t,P,e),s(t,c,e),s(t,D,e),s(t,u,e),s(t,A,e),s(t,b,e),s(t,O,e),s(t,v,e),s(t,I,e),F(w,t,e),s(t,S,e),s(t,M,e),j=!0},p:et,i(t){j||(N(h.$$.fragment,t),N(f.$$.fragment,t),N(w.$$.fragment,t),j=!0)},o(t){R(h.$$.fragment,t),R(f.$$.fragment,t),R(w.$$.fragment,t),j=!1},d(t){t&&(a(C),a(_),a(x),a(L),a(T),a(m),a(k),a(g),a(H),a(y),a(E),a(d),a(P),a(c),a(D),a(u),a(A),a(b),a(O),a(v),a(I),a(S),a(M)),a(r),U(h,t),U(f,t),U(w,t)}}}const ut='{"title":"Create a dataset card","local":"create-a-dataset-card","sections":[],"depth":1}';function pt(G){return at(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class gt extends st{constructor(r){super(),ot(this,r,pt,ct,tt,{})}}export{gt as component};

Xet Storage Details

Size:
6.04 kB
·
Xet hash:
f85e1a8890ed0c5cb760a9557cb8f38e241d439d786b7a501572c60556b9d69e

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.