Buckets:

HuggingFaceDocBuilder's picture
download
raw
12.1 kB
import"../chunks/DsnmJJEf.js";import{i as U,h as X,C as G,H as s,D as e,E as Y,s as Z,a as E}from"../chunks/DT0OpeMJ.js";import{p as D,o as S,s as a,f as q,a as u,b as A,c as t,d as p,n as B,r as n}from"../chunks/Bb-LL0eD.js";import{E as V}from"../chunks/-HcYSbRW.js";const K='{"title":"4-bit quantization","local":"4-bit-quantization","sections":[{"title":"Linear4bit","local":"bitsandbytes.nn.Linear4bit","sections":[],"depth":2},{"title":"LinearFP4","local":"bitsandbytes.nn.LinearFP4","sections":[],"depth":2},{"title":"LinearNF4","local":"bitsandbytes.nn.LinearNF4","sections":[],"depth":2},{"title":"Params4bit","local":"bitsandbytes.nn.Params4bit","sections":[],"depth":2}],"depth":1}';var O=p('<meta name="hf:doc:metadata"/>'),$=p("<p>Example:</p> <!>",1),H=p(`<p></p> <!> <!> <p><a href="https://hf.co/papers/2305.14314" rel="nofollow">QLoRA</a> is a finetuning method that quantizes a model to 4-bits and adds a set of low-rank adaptation (LoRA) weights to the model and tuning them through the quantized weights. This method also introduces a new data type, 4-bit NormalFloat (<code>LinearNF4</code>) in addition to the standard Float4 data type (<code>LinearFP4</code>). <code>LinearNF4</code> is a quantization data type for normally distributed data and can improve performance.</p> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>This class is the base module for the 4-bit quantization algorithm presented in <a href="https://arxiv.org/abs/2305.14314" rel="nofollow">QLoRA</a>.
QLoRA 4-bit linear layers uses blockwise k-bit quantization under the hood, with the possibility of selecting various
compute datatypes such as FP4 and NF4.</p> <p>In order to quantize a linear layer one should first load the original fp16 / bf16 weights into
the Linear4bit module, then call <code>quantized_module.to("cuda")</code> to quantize the fp16 / bf16 weights.</p> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Initialize Linear4bit class.</p></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Implements the FP4 data type.</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <p>Implements the NF4 data type.</p> <p>Constructs a quantization data type where each bin has equal area under a standard normal distribution N(0, 1) that
is normalized into the range [-1, 1].</p> <p>For more information read the paper: QLoRA: Efficient Finetuning of Quantized LLMs (<a href="https://arxiv.org/abs/2305.14314" rel="nofollow">https://arxiv.org/abs/2305.14314</a>)</p> <p>Implementation of the NF4 data type in bitsandbytes can be found in the <code>create_normal_map</code> function in
the <code>functional.py</code> file: <a href="https://github.com/TimDettmers/bitsandbytes/blob/main/bitsandbytes/functional.py#L236" rel="nofollow">https://github.com/TimDettmers/bitsandbytes/blob/main/bitsandbytes/functional.py#L236</a>.</p> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!></div></div> <!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!> <div class="docstring border-l-2 border-t-2 pl-4 pt-3.5 border-gray-100 rounded-tl-xl mb-6 mt-8"><!></div></div> <!> <p></p>`,1);function sa(z,k){D(k,!1),S(()=>{new URLSearchParams(window.location.search).get("fw")}),U();var m=H();X("hikc72",i=>{var b=O();Z(b,"content",K),u(i,b)});var c=a(q(m),2);G(c,{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"});var h=a(c,2);s(h,{title:"4-bit quantization",local:"4-bit-quantization",headingTag:"h1"});var _=a(h,4);s(_,{title:"Linear4bit",local:"bitsandbytes.nn.Linear4bit",headingTag:"h2"});var r=a(_,2),y=t(r);e(y,{name:"class bitsandbytes.nn.Linear4bit",anchor:"bitsandbytes.nn.Linear4bit",source:"https://github.com/bitsandbytes-foundation/bitsandbytes/blob/vr_2013/bitsandbytes/nn/modules.py#L504",parameters:[{name:"input_features",val:""},{name:"output_features",val:""},{name:"bias",val:" = True"},{name:"compute_dtype",val:" = None"},{name:"compress_statistics",val:" = True"},{name:"quant_type",val:" = 'fp4'"},{name:"quant_storage",val:" = torch.uint8"},{name:"device",val:" = None"}]});var v=a(y,6);V(v,{anchor:"bitsandbytes.nn.Linear4bit.example",children:(i,b)=>{var J=$(),C=a(q(J),2);E(C,{code:"aW1wb3J0JTIwdG9yY2glMEFpbXBvcnQlMjB0b3JjaC5ubiUyMGFzJTIwbm4lMEElMEFpbXBvcnQlMjBiaXRzYW5kYnl0ZXMlMjBhcyUyMGJuYiUwQWZyb20lMjBiaXRzYW5kYnl0ZXMubm4lMjBpbXBvcnQlMjBMaW5lYXI0Yml0JTBBJTBBZnAxNl9tb2RlbCUyMCUzRCUyMG5uLlNlcXVlbnRpYWwoJTBBJTIwJTIwJTIwJTIwbm4uTGluZWFyKDY0JTJDJTIwNjQpJTJDJTBBJTIwJTIwJTIwJTIwbm4uTGluZWFyKDY0JTJDJTIwNjQpJTBBKSUwQSUwQXF1YW50aXplZF9tb2RlbCUyMCUzRCUyMG5uLlNlcXVlbnRpYWwoJTBBJTIwJTIwJTIwJTIwTGluZWFyNGJpdCg2NCUyQyUyMDY0KSUyQyUwQSUyMCUyMCUyMCUyMExpbmVhcjRiaXQoNjQlMkMlMjA2NCklMEEpJTBBJTBBcXVhbnRpemVkX21vZGVsLmxvYWRfc3RhdGVfZGljdChmcDE2X21vZGVsLnN0YXRlX2RpY3QoKSklMEFxdWFudGl6ZWRfbW9kZWwlMjAlM0QlMjBxdWFudGl6ZWRfbW9kZWwudG8oMCklMjAlMjMlMjBRdWFudGl6YXRpb24lMjBoYXBwZW5zJTIwaGVyZQ==",highlighted:`<span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">import</span> torch.nn <span class="hljs-keyword">as</span> nn
<span class="hljs-keyword">import</span> bitsandbytes <span class="hljs-keyword">as</span> bnb
<span class="hljs-keyword">from</span> bitsandbytes.nn <span class="hljs-keyword">import</span> Linear4bit
fp16_model = nn.Sequential(
nn.Linear(<span class="hljs-number">64</span>, <span class="hljs-number">64</span>),
nn.Linear(<span class="hljs-number">64</span>, <span class="hljs-number">64</span>)
)
quantized_model = nn.Sequential(
Linear4bit(<span class="hljs-number">64</span>, <span class="hljs-number">64</span>),
Linear4bit(<span class="hljs-number">64</span>, <span class="hljs-number">64</span>)
)
quantized_model.load_state_dict(fp16_model.state_dict())
quantized_model = quantized_model.to(<span class="hljs-number">0</span>) <span class="hljs-comment"># Quantization happens here</span>`,lang:"python",wrap:!1}),u(i,J)},$$slots:{default:!0}});var f=a(v,2),R=t(f);e(R,{name:"__init__",anchor:"bitsandbytes.nn.Linear4bit.__init__",source:"https://github.com/bitsandbytes-foundation/bitsandbytes/blob/vr_2013/bitsandbytes/nn/modules.py#L537",parameters:[{name:"input_features",val:""},{name:"output_features",val:""},{name:"bias",val:" = True"},{name:"compute_dtype",val:" = None"},{name:"compress_statistics",val:" = True"},{name:"quant_type",val:" = 'fp4'"},{name:"quant_storage",val:" = torch.uint8"},{name:"device",val:" = None"}],parametersDescription:[{anchor:"bitsandbytes.nn.Linear4bit.__init__.input_features",description:`<strong>input_features</strong> (<code>str</code>) &#x2014;
Number of input features of the linear layer.`,name:"input_features"},{anchor:"bitsandbytes.nn.Linear4bit.__init__.output_features",description:`<strong>output_features</strong> (<code>str</code>) &#x2014;
Number of output features of the linear layer.`,name:"output_features"},{anchor:"bitsandbytes.nn.Linear4bit.__init__.bias",description:`<strong>bias</strong> (<code>bool</code>, defaults to <code>True</code>) &#x2014;
Whether the linear class uses the bias term as well.`,name:"bias"}]}),B(2),n(f),n(r);var g=a(r,2);s(g,{title:"LinearFP4",local:"bitsandbytes.nn.LinearFP4",headingTag:"h2"});var o=a(g,2),L=t(o);e(L,{name:"class bitsandbytes.nn.LinearFP4",anchor:"bitsandbytes.nn.LinearFP4",source:"https://github.com/bitsandbytes-foundation/bitsandbytes/blob/vr_2013/bitsandbytes/nn/modules.py#L640",parameters:[{name:"input_features",val:""},{name:"output_features",val:""},{name:"bias",val:" = True"},{name:"compute_dtype",val:" = None"},{name:"compress_statistics",val:" = True"},{name:"quant_storage",val:" = torch.uint8"},{name:"device",val:" = None"}]});var w=a(L,4),W=t(w);e(W,{name:"__init__",anchor:"bitsandbytes.nn.LinearFP4.__init__",source:"https://github.com/bitsandbytes-foundation/bitsandbytes/blob/vr_2013/bitsandbytes/nn/modules.py#L645",parameters:[{name:"input_features",val:""},{name:"output_features",val:""},{name:"bias",val:" = True"},{name:"compute_dtype",val:" = None"},{name:"compress_statistics",val:" = True"},{name:"quant_storage",val:" = torch.uint8"},{name:"device",val:" = None"}],parametersDescription:[{anchor:"bitsandbytes.nn.LinearFP4.__init__.input_features",description:`<strong>input_features</strong> (<code>str</code>) &#x2014;
Number of input features of the linear layer.`,name:"input_features"},{anchor:"bitsandbytes.nn.LinearFP4.__init__.output_features",description:`<strong>output_features</strong> (<code>str</code>) &#x2014;
Number of output features of the linear layer.`,name:"output_features"},{anchor:"bitsandbytes.nn.LinearFP4.__init__.bias",description:`<strong>bias</strong> (<code>bool</code>, defaults to <code>True</code>) &#x2014;
Whether the linear class uses the bias term as well.`,name:"bias"}]}),n(w),n(o);var T=a(o,2);s(T,{title:"LinearNF4",local:"bitsandbytes.nn.LinearNF4",headingTag:"h2"});var l=a(T,2),N=t(l);e(N,{name:"class bitsandbytes.nn.LinearNF4",anchor:"bitsandbytes.nn.LinearNF4",source:"https://github.com/bitsandbytes-foundation/bitsandbytes/blob/vr_2013/bitsandbytes/nn/modules.py#L676",parameters:[{name:"input_features",val:""},{name:"output_features",val:""},{name:"bias",val:" = True"},{name:"compute_dtype",val:" = None"},{name:"compress_statistics",val:" = True"},{name:"quant_storage",val:" = torch.uint8"},{name:"device",val:" = None"}]});var F=a(N,10),P=t(F);e(P,{name:"__init__",anchor:"bitsandbytes.nn.LinearNF4.__init__",source:"https://github.com/bitsandbytes-foundation/bitsandbytes/blob/vr_2013/bitsandbytes/nn/modules.py#L688",parameters:[{name:"input_features",val:""},{name:"output_features",val:""},{name:"bias",val:" = True"},{name:"compute_dtype",val:" = None"},{name:"compress_statistics",val:" = True"},{name:"quant_storage",val:" = torch.uint8"},{name:"device",val:" = None"}],parametersDescription:[{anchor:"bitsandbytes.nn.LinearNF4.__init__.input_features",description:`<strong>input_features</strong> (<code>str</code>) &#x2014;
Number of input features of the linear layer.`,name:"input_features"},{anchor:"bitsandbytes.nn.LinearNF4.__init__.output_features",description:`<strong>output_features</strong> (<code>str</code>) &#x2014;
Number of output features of the linear layer.`,name:"output_features"},{anchor:"bitsandbytes.nn.LinearNF4.__init__.bias",description:`<strong>bias</strong> (<code>bool</code>, defaults to <code>True</code>) &#x2014;
Whether the linear class uses the bias term as well.`,name:"bias"}]}),n(F),n(l);var M=a(l,2);s(M,{title:"Params4bit",local:"bitsandbytes.nn.Params4bit",headingTag:"h2"});var d=a(M,2),j=t(d);e(j,{name:"class bitsandbytes.nn.Params4bit",anchor:"bitsandbytes.nn.Params4bit",source:"https://github.com/bitsandbytes-foundation/bitsandbytes/blob/vr_2013/bitsandbytes/nn/modules.py#L213",parameters:[{name:"data",val:": typing.Optional[torch.Tensor] = None"},{name:"requires_grad",val:" = False"},{name:"quant_state",val:": typing.Optional[bitsandbytes.functional.QuantState] = None"},{name:"blocksize",val:": typing.Optional[int] = None"},{name:"compress_statistics",val:": bool = True"},{name:"quant_type",val:": str = 'fp4'"},{name:"quant_storage",val:": dtype = torch.uint8"},{name:"module",val:": typing.Optional[ForwardRef('Linear4bit')] = None"},{name:"bnb_quantized",val:": bool = False"},{name:"**kwargs",val:""}]});var x=a(j,2),I=t(x);e(I,{name:"<lambda>",anchor:"bitsandbytes.nn.Params4bit.__init__",source:"https://github.com/bitsandbytes-foundation/bitsandbytes/blob/vr_2013/doc_builder/mock_imports.py#L251",parameters:[{name:"*args",val:""},{name:"**kwargs",val:""}]}),n(x),n(d);var Q=a(d,2);Y(Q,{source:"https://github.com/bitsandbytes-foundation/bitsandbytes/blob/main/docs/source/reference/nn/linear4bit.mdx"}),B(2),u(z,m),A()}export{sa as component};

Xet Storage Details

Size:
12.1 kB
·
Xet hash:
c7d384666dd56c0c0e82c32f3c5ed51861cdc4e32cc90c1a39c8bf81b1e07c1e

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.