Buckets:

HuggingFaceDocBuilder's picture
download
raw
18.9 kB
import{s as he,n as ye,o as we}from"../chunks/scheduler.56725da7.js";import{S as je,i as Te,e as o,s,c as m,h as Je,a as i,d as n,b as a,f as ce,g as d,j as b,k as be,l as fe,m as t,n as M,t as p,o as r,p as u}from"../chunks/index.18a26576.js";import{C as Ue}from"../chunks/CopyLLMTxtMenu.a1f2bcd7.js";import{C as se}from"../chunks/CodeBlock.d6d1e300.js";import{H as R}from"../chunks/MermaidChart.svelte_svelte_type_style_lang.9f98faf7.js";function Ze(ae){let c,Q,X,C,h,z,y,N,w,oe="This guide explains how to convert, load, and use Qwen Embedding models on AWS Trainium and Inferentia2 using Optimum Neuron. The Qwen3 Embedding model series is the latest proprietary model of the Qwen family, specifically designed for text embedding and ranking tasks. Building upon the dense foundational models of the Qwen3 series, it provides a comprehensive range of text embeddings and reranking models in various sizes (0.6B, 4B, and 8B). The Qwen3 Embedding series offer support for over 100 languages, thanks to the multilingual capabilites of Qwen3 models.",F,j,A,T,ie='You can run this notebook on a AWS EC2 instance with the HF DLAMI. To create an instance with the DLAMI, you can follow the <a href="https://huggingface.co/docs/optimum-neuron/en/ec2-setup" rel="nofollow">EC2 Setup guide</a>. Alternatively if you are on a AWS Trainium and Inferentia instance, you can manually install <code>optimum-neuron</code> using the steps in the <a href="https://huggingface.co/docs/optimum-neuron/en/ec2-setup#alternative-manual-installation" rel="nofollow">manual installation guide</a>.',$,J,me="This guide is written using a <code>trn2.3xlarge</code> AWS Trainium2 instance. But you can use the same code to run the model using a AWS Inferentia2 instance like <code>inf2.48xlarge</code>.",S,f,x,U,de="First, you need to convert the model to a format compatible with AWS Trainium and Inferentia2. You can compile Qwen3 Embedding models with Optimum Neuron using the <code>optimum-cli</code> or <code>NeuronModelForEmbedding</code> class. Below you will find an example for both approaches.",H,Z,Me='In the below example, we illustrate this using <a href="https://huggingface.co/Qwen/Qwen3-Embedding-8B" rel="nofollow">Qwen3 Embedding 8B</a> but you can follow the same steps for the 0.8B and the 4B versions of the embedding models.',Y,g,q,B,pe="Here we will use the <code>NeuronModelForEmbedding</code> class, which can convert Qwen3 Embedding models to a format compatible with AWS Trainium and Inferentia2 or load already converted models. When exporting models with <code>NeuronModelForEmbedding</code> you need to define the the <code>sequence_length</code> and <code>batch size</code> in the neuron config.",L,W,D,G,P,I,re="Here we will use the <code>optimum-cli</code> tool to convert the model. Similar to the <code>NeuronModelForEmbedding</code> we need to define our sequence length and batch size. The <code>optimum-cli</code> will automatically convert the model to a format compatible with AWS Trainium and Inferentia2 and save it to the specified output directory.",O,E,K,V,ee,_,ue=`Once we have a compiled the model, for loading the model we can use the <code>NeuronModelForEmbedding</code> class to load the model and run inference.
In the below example, we first compute embeddings for two queries and documents; and then compute the similarity score for the queries and documents.`,le,k,ne,v,te;return h=new Ue({props:{containerStyle:"float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"}}),y=new R({props:{title:"Qwen3 Embedding on AWS Trainium with Optimum Neuron",local:"qwen3-embedding-on-aws-trainium-with-optimum-neuron",headingTag:"h1"}}),j=new R({props:{title:"Prerequisite: Setup Environment",local:"prerequisite-setup-environment",headingTag:"h2"}}),f=new R({props:{title:"Compile Qwen Embedding Models for AWS Trainium",local:"compile-qwen-embedding-models-for-aws-trainium",headingTag:"h2"}}),g=new R({props:{title:"Option A: Compile using the NeuronModelForEmbedding class",local:"option-a-compile-using-the-neuronmodelforembedding-class",headingTag:"h3"}}),W=new se({props:{code:"ZnJvbSUyMG9wdGltdW0ubmV1cm9uJTIwaW1wb3J0JTIwTmV1cm9uTW9kZWxGb3JFbWJlZGRpbmclMEElMEFtb2RlbF9pZCUyMCUzRCUyMCUyMlF3ZW4lMkZRd2VuMy1FbWJlZGRpbmctOEIlMjIlMEFuZXVyb25fbW9kZWxfZGlyJTIwJTNEJTIwJTIycXdlbl9lbWJlZGRpbmdfOEJfdHA0JTIyJTBBJTBBJTIzJTIwSWYlMjB5b3UlMjBhcmUlMjB1c2luZyUyMGElMjBBV1MlMjBJbmZlcmVudGlhMiUyMGluc3RhbmNlJTIwYW5kJTIwdXNlJTIwJ3RlbnNvcl9wYXJhbGxlbF9zaXplJTNENCclMkMlMjB5b3UlMjBzaG91bGQlMjBzZXQlMjB0aGUlMjBmb2xsb3dpbmclMjBlbnZpcm9ubWVudCUyMHZhcmlhYmxlJTIwYXMlMjB3ZWxsLiUwQSUyMyUyMGltcG9ydCUyMG9zJTBBJTIzJTIwb3MuZW52aXJvbiU1QiUyMkxPQ0FMX1dPUkxEX1NJWkUlMjIlNUQlMjAlM0QlMjAnNCclMEElMEFuZXVyb25fY29uZmlnJTIwJTNEJTIwTmV1cm9uTW9kZWxGb3JFbWJlZGRpbmcuZ2V0X25ldXJvbl9jb25maWcoJTBBJTIwJTIwJTIwJTIwbW9kZWxfaWQlMkMlMjBiYXRjaF9zaXplJTNEMiUyQyUyMHNlcXVlbmNlX2xlbmd0aCUzRDEwMjQlMkMlMjB0ZW5zb3JfcGFyYWxsZWxfc2l6ZSUzRDQlMEEpJTBBJTBBbmV1cm9uX21vZGVsJTIwJTNEJTIwTmV1cm9uTW9kZWxGb3JFbWJlZGRpbmcuZXhwb3J0KG1vZGVsX2lkJTNEbW9kZWxfaWQlMkMlMjBuZXVyb25fY29uZmlnJTNEbmV1cm9uX2NvbmZpZyUyQyUyMGxvYWRfd2VpZ2h0cyUzREZhbHNlKSUwQSUwQSUyMyUyMFNhdmUlMjBtb2RlbCUyMHRvJTIwZGlzayUwQW5ldXJvbl9tb2RlbC5zYXZlX3ByZXRyYWluZWQobmV1cm9uX21vZGVsX2Rpcik=",highlighted:`<span class="hljs-keyword">from</span> optimum.neuron <span class="hljs-keyword">import</span> NeuronModelForEmbedding
model_id = <span class="hljs-string">&quot;Qwen/Qwen3-Embedding-8B&quot;</span>
neuron_model_dir = <span class="hljs-string">&quot;qwen_embedding_8B_tp4&quot;</span>
<span class="hljs-comment"># If you are using a AWS Inferentia2 instance and use &#x27;tensor_parallel_size=4&#x27;, you should set the following environment variable as well.</span>
<span class="hljs-comment"># import os</span>
<span class="hljs-comment"># os.environ[&quot;LOCAL_WORLD_SIZE&quot;] = &#x27;4&#x27;</span>
neuron_config = NeuronModelForEmbedding.get_neuron_config(
model_id, batch_size=<span class="hljs-number">2</span>, sequence_length=<span class="hljs-number">1024</span>, tensor_parallel_size=<span class="hljs-number">4</span>
)
neuron_model = NeuronModelForEmbedding.export(model_id=model_id, neuron_config=neuron_config, load_weights=<span class="hljs-literal">False</span>)
<span class="hljs-comment"># Save model to disk</span>
neuron_model.save_pretrained(neuron_model_dir)`,lang:"python",wrap:!1}}),G=new R({props:{title:"Option B: Compile using the optimum-cli tool",local:"option-b-compile-using-the-optimum-cli-tool",headingTag:"h3"}}),E=new se({props:{code:"ISUyMG9wdGltdW0tY2xpJTIwZXhwb3J0JTIwbmV1cm9uJTIwLS1tb2RlbCUyMFF3ZW4lMkZRd2VuMy1FbWJlZGRpbmctOEIlMjAtLWJhdGNoX3NpemUlMjAyJTIwLS1zZXF1ZW5jZV9sZW5ndGglMjAxMDI0JTIwLS1hdXRvX2Nhc3QlMjBtYXRtdWwlMjAtLWluc3RhbmNlX3R5cGUlMjB0cm4yJTIwLS10ZW5zb3JfcGFyYWxsZWxfc2l6ZSUyMDQlMjBxd2VuX2VtYmVkZGluZ184Ql90cDQlMkY=",highlighted:'! optimum-cli export neuron --model Qwen/Qwen3-Embedding-8B --batch_size <span class="hljs-number">2</span> --sequence_length <span class="hljs-number">1024</span> --auto_cast matmul --instance_type trn2 --tensor_parallel_size <span class="hljs-number">4</span> qwen_embedding_8B_tp4/',lang:"python",wrap:!1}}),V=new R({props:{title:"Load compiled Qwen3 Embedding model and run inference",local:"load-compiled-qwen3-embedding-model-and-run-inference",headingTag:"h2"}}),k=new se({props:{code:"aW1wb3J0JTIwdG9yY2glMEFmcm9tJTIwdG9yY2glMjBpbXBvcnQlMjBUZW5zb3IlMEFpbXBvcnQlMjB0b3JjaC5ubi5mdW5jdGlvbmFsJTIwYXMlMjBGJTBBJTBBZnJvbSUyMHRyYW5zZm9ybWVycyUyMGltcG9ydCUyMEF1dG9Ub2tlbml6ZXIlMEFmcm9tJTIwb3B0aW11bS5uZXVyb24lMjBpbXBvcnQlMjBOZXVyb25Nb2RlbEZvckVtYmVkZGluZyUwQSUwQSUyMyUyMHNldCUyMHRoaXMlMjB0byUyMHRoZSUyMHBhdGglMjB0aGF0JTIwaGFzJTIwdGhlJTIwY29tcGlsZWQlMjBtb2RlbCUyMGZyb20lMjBhYm92ZSUwQW1vZGVsX2lkX29yX3BhdGglMjAlM0QlMjBuZXVyb25fbW9kZWxfZGlyJTBBJTBBJTBBZGVmJTIwbGFzdF90b2tlbl9wb29sKGxhc3RfaGlkZGVuX3N0YXRlcyUzQSUyMFRlbnNvciUyQyUyMGF0dGVudGlvbl9tYXNrJTNBJTIwVGVuc29yKSUyMC0lM0UlMjBUZW5zb3IlM0ElMEElMjAlMjAlMjAlMjBsZWZ0X3BhZGRpbmclMjAlM0QlMjBhdHRlbnRpb25fbWFzayU1QiUzQSUyQyUyMC0xJTVELnN1bSgpJTIwJTNEJTNEJTIwYXR0ZW50aW9uX21hc2suc2hhcGUlNUIwJTVEJTBBJTIwJTIwJTIwJTIwaWYlMjBsZWZ0X3BhZGRpbmclM0ElMEElMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjByZXR1cm4lMjBsYXN0X2hpZGRlbl9zdGF0ZXMlNUIlM0ElMkMlMjAtMSU1RCUwQSUyMCUyMCUyMCUyMGVsc2UlM0ElMEElMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjBzZXF1ZW5jZV9sZW5ndGhzJTIwJTNEJTIwYXR0ZW50aW9uX21hc2suc3VtKGRpbSUzRDEpJTIwLSUyMDElMEElMjAlMjAlMjAlMjAlMjAlMjAlMjAlMjBiYXRjaF9zaXplJTIwJTNEJTIwbGFzdF9oaWRkZW5fc3RhdGVzLnNoYXBlJTVCMCU1RCUwQSUyMCUyMCUyMCUyMCUyMCUyMCUyMCUyMHJldHVybiUyMGxhc3RfaGlkZGVuX3N0YXRlcyU1QnRvcmNoLmFyYW5nZShiYXRjaF9zaXplJTJDJTIwZGV2aWNlJTNEbGFzdF9oaWRkZW5fc3RhdGVzLmRldmljZSklMkMlMjBzZXF1ZW5jZV9sZW5ndGhzJTVEJTBBJTBBJTBBJTIzJTIwTG9hZCUyMG1vZGVsJTIwYW5kJTIwdG9rZW5pemVyJTBBbW9kZWwlMjAlM0QlMjBOZXVyb25Nb2RlbEZvckVtYmVkZGluZy5mcm9tX3ByZXRyYWluZWQobW9kZWxfaWRfb3JfcGF0aCklMEF0b2tlbml6ZXIlMjAlM0QlMjBBdXRvVG9rZW5pemVyLmZyb21fcHJldHJhaW5lZChtb2RlbF9pZCUyQyUyMHBhZGRpbmdfc2lkZSUzRCUyMnJpZ2h0JTIyKSUwQSUwQSUyMyUyMElucHV0JTIwdGV4dCUyMHRvJTIwZW1iZWQlMEFxdWVyaWVzJTIwJTNEJTIwJTVCJTIyV2hhdCUyMGlzJTIwdGhlJTIwY2FwaXRhbCUyMG9mJTIwQ2hpbmElM0YlMjIlMkMlMjAlMjJFeHBsYWluJTIwZ3Jhdml0eSUyMiU1RCUwQSUyMyUyME5vJTIwbmVlZCUyMHRvJTIwYWRkJTIwaW5zdHJ1Y3Rpb24lMjBmb3IlMjByZXRyaWV2YWwlMjBkb2N1bWVudHMlMEFkb2N1bWVudHMlMjAlM0QlMjAlNUIlMEElMjAlMjAlMjAlMjAlMjJUaGUlMjBjYXBpdGFsJTIwb2YlMjBDaGluYSUyMGlzJTIwQmVpamluZy4lMjIlMkMlMEElMjAlMjAlMjAlMjAlMjJHcmF2aXR5JTIwaXMlMjBhJTIwZm9yY2UlMjB0aGF0JTIwYXR0cmFjdHMlMjB0d28lMjBib2RpZXMlMjB0b3dhcmRzJTIwZWFjaCUyMG90aGVyLiUyMEl0JTIwZ2l2ZXMlMjB3ZWlnaHQlMjB0byUyMHBoeXNpY2FsJTIwb2JqZWN0cyUyMGFuZCUyMGlzJTIwcmVzcG9uc2libGUlMjBmb3IlMjB0aGUlMjBtb3ZlbWVudCUyMG9mJTIwcGxhbmV0cyUyMGFyb3VuZCUyMHRoZSUyMHN1bi4lMjIlMkMlMEElNUQlMEElMEElMEElMjMlMjBUb2tlbml6ZSUyMHRoZSUyMGlucHV0JTIwdGV4dHMlMEFxdWVyaWVzX3Rva2VucyUyMCUzRCUyMHRva2VuaXplciglMEElMjAlMjAlMjAlMjBxdWVyaWVzJTJDJTBBJTIwJTIwJTIwJTIwcGFkZGluZyUzRFRydWUlMkMlMEElMjAlMjAlMjAlMjB0cnVuY2F0aW9uJTNEVHJ1ZSUyQyUwQSUyMCUyMCUyMCUyMG1heF9sZW5ndGglM0Q4MTkyJTJDJTBBJTIwJTIwJTIwJTIwcmV0dXJuX3RlbnNvcnMlM0QlMjJwdCUyMiUyQyUwQSklMEFkb2N1bWVudHNfdG9rZW5zJTIwJTNEJTIwdG9rZW5pemVyKCUwQSUyMCUyMCUyMCUyMGRvY3VtZW50cyUyQyUwQSUyMCUyMCUyMCUyMHBhZGRpbmclM0RUcnVlJTJDJTBBJTIwJTIwJTIwJTIwdHJ1bmNhdGlvbiUzRFRydWUlMkMlMEElMjAlMjAlMjAlMjBtYXhfbGVuZ3RoJTNEODE5MiUyQyUwQSUyMCUyMCUyMCUyMHJldHVybl90ZW5zb3JzJTNEJTIycHQlMjIlMkMlMEEpJTBBJTBBb3V0cHV0cyUyMCUzRCUyMG1vZGVsKCoqcXVlcmllc190b2tlbnMpJTBBcXVlcmllc19lbWJlZGRpbmdzJTIwJTNEJTIwbGFzdF90b2tlbl9wb29sKG91dHB1dHMlMkMlMjBxdWVyaWVzX3Rva2VucyU1QiUyMmF0dGVudGlvbl9tYXNrJTIyJTVEKSUwQSUwQW91dHB1dHMlMjAlM0QlMjBtb2RlbCgqKmRvY3VtZW50c190b2tlbnMpJTBBZG9jdW1lbnRzX2VtYmVkZGluZ3MlMjAlM0QlMjBsYXN0X3Rva2VuX3Bvb2wob3V0cHV0cyUyQyUyMGRvY3VtZW50c190b2tlbnMlNUIlMjJhdHRlbnRpb25fbWFzayUyMiU1RCklMEElMEElMjMlMjBub3JtYWxpemUlMjBlbWJlZGRpbmdzJTIwYW5kJTIwY29tcHV0ZSUyMHNpbWlsYXJpdHklMjBzY29yZXMlMEFxdWVyaWVzX2VtYmVkZGluZ3MlMjAlM0QlMjBGLm5vcm1hbGl6ZShxdWVyaWVzX2VtYmVkZGluZ3MlMkMlMjBwJTNEMiUyQyUyMGRpbSUzRDEpJTBBZG9jdW1lbnRzX2VtYmVkZGluZ3MlMjAlM0QlMjBGLm5vcm1hbGl6ZShkb2N1bWVudHNfZW1iZWRkaW5ncyUyQyUyMHAlM0QyJTJDJTIwZGltJTNEMSklMEElMEFzY29yZXMlMjAlM0QlMjBxdWVyaWVzX2VtYmVkZGluZ3MlMjAlNDAlMjBkb2N1bWVudHNfZW1iZWRkaW5ncy5UJTBBcHJpbnQoJTIyU2ltaWxhcml0eSUyMFNjb3JlcyUyMC0tJTNFJTIyJTJDJTIwc2NvcmVzLnRvbGlzdCgpKQ==",highlighted:`<span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">from</span> torch <span class="hljs-keyword">import</span> Tensor
<span class="hljs-keyword">import</span> torch.nn.functional <span class="hljs-keyword">as</span> F
<span class="hljs-keyword">from</span> transformers <span class="hljs-keyword">import</span> AutoTokenizer
<span class="hljs-keyword">from</span> optimum.neuron <span class="hljs-keyword">import</span> NeuronModelForEmbedding
<span class="hljs-comment"># set this to the path that has the compiled model from above</span>
model_id_or_path = neuron_model_dir
<span class="hljs-keyword">def</span> <span class="hljs-title function_">last_token_pool</span>(<span class="hljs-params">last_hidden_states: Tensor, attention_mask: Tensor</span>) -&gt; Tensor:
left_padding = attention_mask[:, -<span class="hljs-number">1</span>].<span class="hljs-built_in">sum</span>() == attention_mask.shape[<span class="hljs-number">0</span>]
<span class="hljs-keyword">if</span> left_padding:
<span class="hljs-keyword">return</span> last_hidden_states[:, -<span class="hljs-number">1</span>]
<span class="hljs-keyword">else</span>:
sequence_lengths = attention_mask.<span class="hljs-built_in">sum</span>(dim=<span class="hljs-number">1</span>) - <span class="hljs-number">1</span>
batch_size = last_hidden_states.shape[<span class="hljs-number">0</span>]
<span class="hljs-keyword">return</span> last_hidden_states[torch.arange(batch_size, device=last_hidden_states.device), sequence_lengths]
<span class="hljs-comment"># Load model and tokenizer</span>
model = NeuronModelForEmbedding.from_pretrained(model_id_or_path)
tokenizer = AutoTokenizer.from_pretrained(model_id, padding_side=<span class="hljs-string">&quot;right&quot;</span>)
<span class="hljs-comment"># Input text to embed</span>
queries = [<span class="hljs-string">&quot;What is the capital of China?&quot;</span>, <span class="hljs-string">&quot;Explain gravity&quot;</span>]
<span class="hljs-comment"># No need to add instruction for retrieval documents</span>
documents = [
<span class="hljs-string">&quot;The capital of China is Beijing.&quot;</span>,
<span class="hljs-string">&quot;Gravity is a force that attracts two bodies towards each other. It gives weight to physical objects and is responsible for the movement of planets around the sun.&quot;</span>,
]
<span class="hljs-comment"># Tokenize the input texts</span>
queries_tokens = tokenizer(
queries,
padding=<span class="hljs-literal">True</span>,
truncation=<span class="hljs-literal">True</span>,
max_length=<span class="hljs-number">8192</span>,
return_tensors=<span class="hljs-string">&quot;pt&quot;</span>,
)
documents_tokens = tokenizer(
documents,
padding=<span class="hljs-literal">True</span>,
truncation=<span class="hljs-literal">True</span>,
max_length=<span class="hljs-number">8192</span>,
return_tensors=<span class="hljs-string">&quot;pt&quot;</span>,
)
outputs = model(**queries_tokens)
queries_embeddings = last_token_pool(outputs, queries_tokens[<span class="hljs-string">&quot;attention_mask&quot;</span>])
outputs = model(**documents_tokens)
documents_embeddings = last_token_pool(outputs, documents_tokens[<span class="hljs-string">&quot;attention_mask&quot;</span>])
<span class="hljs-comment"># normalize embeddings and compute similarity scores</span>
queries_embeddings = F.normalize(queries_embeddings, p=<span class="hljs-number">2</span>, dim=<span class="hljs-number">1</span>)
documents_embeddings = F.normalize(documents_embeddings, p=<span class="hljs-number">2</span>, dim=<span class="hljs-number">1</span>)
scores = queries_embeddings @ documents_embeddings.T
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;Similarity Scores --&gt;&quot;</span>, scores.tolist())`,lang:"python",wrap:!1}}),{c(){c=o("meta"),Q=s(),X=o("p"),C=s(),m(h.$$.fragment),z=s(),m(y.$$.fragment),N=s(),w=o("p"),w.textContent=oe,F=s(),m(j.$$.fragment),A=s(),T=o("p"),T.innerHTML=ie,$=s(),J=o("p"),J.innerHTML=me,S=s(),m(f.$$.fragment),x=s(),U=o("p"),U.innerHTML=de,H=s(),Z=o("p"),Z.innerHTML=Me,Y=s(),m(g.$$.fragment),q=s(),B=o("p"),B.innerHTML=pe,L=s(),m(W.$$.fragment),D=s(),m(G.$$.fragment),P=s(),I=o("p"),I.innerHTML=re,O=s(),m(E.$$.fragment),K=s(),m(V.$$.fragment),ee=s(),_=o("p"),_.innerHTML=ue,le=s(),m(k.$$.fragment),ne=s(),v=o("p"),this.h()},l(e){const l=Je("svelte-u9bgzb",document.head);c=i(l,"META",{name:!0,content:!0}),l.forEach(n),Q=a(e),X=i(e,"P",{}),ce(X).forEach(n),C=a(e),d(h.$$.fragment,e),z=a(e),d(y.$$.fragment,e),N=a(e),w=i(e,"P",{"data-svelte-h":!0}),b(w)!=="svelte-1wd3ojk"&&(w.textContent=oe),F=a(e),d(j.$$.fragment,e),A=a(e),T=i(e,"P",{"data-svelte-h":!0}),b(T)!=="svelte-1yrpmjk"&&(T.innerHTML=ie),$=a(e),J=i(e,"P",{"data-svelte-h":!0}),b(J)!=="svelte-hpbov5"&&(J.innerHTML=me),S=a(e),d(f.$$.fragment,e),x=a(e),U=i(e,"P",{"data-svelte-h":!0}),b(U)!=="svelte-1g2aewj"&&(U.innerHTML=de),H=a(e),Z=i(e,"P",{"data-svelte-h":!0}),b(Z)!=="svelte-eh7wv6"&&(Z.innerHTML=Me),Y=a(e),d(g.$$.fragment,e),q=a(e),B=i(e,"P",{"data-svelte-h":!0}),b(B)!=="svelte-1t2eef9"&&(B.innerHTML=pe),L=a(e),d(W.$$.fragment,e),D=a(e),d(G.$$.fragment,e),P=a(e),I=i(e,"P",{"data-svelte-h":!0}),b(I)!=="svelte-18r2uoc"&&(I.innerHTML=re),O=a(e),d(E.$$.fragment,e),K=a(e),d(V.$$.fragment,e),ee=a(e),_=i(e,"P",{"data-svelte-h":!0}),b(_)!=="svelte-1uu32wm"&&(_.innerHTML=ue),le=a(e),d(k.$$.fragment,e),ne=a(e),v=i(e,"P",{}),ce(v).forEach(n),this.h()},h(){be(c,"name","hf:doc:metadata"),be(c,"content",ge)},m(e,l){fe(document.head,c),t(e,Q,l),t(e,X,l),t(e,C,l),M(h,e,l),t(e,z,l),M(y,e,l),t(e,N,l),t(e,w,l),t(e,F,l),M(j,e,l),t(e,A,l),t(e,T,l),t(e,$,l),t(e,J,l),t(e,S,l),M(f,e,l),t(e,x,l),t(e,U,l),t(e,H,l),t(e,Z,l),t(e,Y,l),M(g,e,l),t(e,q,l),t(e,B,l),t(e,L,l),M(W,e,l),t(e,D,l),M(G,e,l),t(e,P,l),t(e,I,l),t(e,O,l),M(E,e,l),t(e,K,l),M(V,e,l),t(e,ee,l),t(e,_,l),t(e,le,l),M(k,e,l),t(e,ne,l),t(e,v,l),te=!0},p:ye,i(e){te||(p(h.$$.fragment,e),p(y.$$.fragment,e),p(j.$$.fragment,e),p(f.$$.fragment,e),p(g.$$.fragment,e),p(W.$$.fragment,e),p(G.$$.fragment,e),p(E.$$.fragment,e),p(V.$$.fragment,e),p(k.$$.fragment,e),te=!0)},o(e){r(h.$$.fragment,e),r(y.$$.fragment,e),r(j.$$.fragment,e),r(f.$$.fragment,e),r(g.$$.fragment,e),r(W.$$.fragment,e),r(G.$$.fragment,e),r(E.$$.fragment,e),r(V.$$.fragment,e),r(k.$$.fragment,e),te=!1},d(e){e&&(n(Q),n(X),n(C),n(z),n(N),n(w),n(F),n(A),n(T),n($),n(J),n(S),n(x),n(U),n(H),n(Z),n(Y),n(q),n(B),n(L),n(D),n(P),n(I),n(O),n(K),n(ee),n(_),n(le),n(ne),n(v)),n(c),u(h,e),u(y,e),u(j,e),u(f,e),u(g,e),u(W,e),u(G,e),u(E,e),u(V,e),u(k,e)}}}const ge='{"title":"Qwen3 Embedding on AWS Trainium with Optimum Neuron","local":"qwen3-embedding-on-aws-trainium-with-optimum-neuron","sections":[{"title":"Prerequisite: Setup Environment","local":"prerequisite-setup-environment","sections":[],"depth":2},{"title":"Compile Qwen Embedding Models for AWS Trainium","local":"compile-qwen-embedding-models-for-aws-trainium","sections":[{"title":"Option A: Compile using the NeuronModelForEmbedding class","local":"option-a-compile-using-the-neuronmodelforembedding-class","sections":[],"depth":3},{"title":"Option B: Compile using the optimum-cli tool","local":"option-b-compile-using-the-optimum-cli-tool","sections":[],"depth":3}],"depth":2},{"title":"Load compiled Qwen3 Embedding model and run inference","local":"load-compiled-qwen3-embedding-model-and-run-inference","sections":[],"depth":2}],"depth":1}';function Be(ae){return we(()=>{new URLSearchParams(window.location.search).get("fw")}),[]}class _e extends je{constructor(c){super(),Te(this,c,Be,Ze,he,{})}}export{_e as component};

Xet Storage Details

Size:
18.9 kB
·
Xet hash:
024532786168744f1c8b6c42588db88b11bcbfe21c41cfffcced40e72dc7663a

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.