Buckets:

hf-doc-build/doc-dev / optimum-neuron /pr_1113 /en /inference_tutorials /compare-book-translations.html
download
raw
94.5 kB
<meta charset="utf-8" /><meta name="hf:doc:metadata" content="{&quot;title&quot;:&quot;Comparing Book Translations using Embeddings&quot;,&quot;local&quot;:&quot;comparing-book-translations-using-embeddings&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Background: Embedding Models&quot;,&quot;local&quot;:&quot;background-embedding-models&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;What This Example Illustrates&quot;,&quot;local&quot;:&quot;what-this-example-illustrates&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Dataset&quot;,&quot;local&quot;:&quot;dataset&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Workflow&quot;,&quot;local&quot;:&quot;workflow&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;1. Configure Inference Endpoint&quot;,&quot;local&quot;:&quot;1-configure-inference-endpoint&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;2. Setup and Dependencies&quot;,&quot;local&quot;:&quot;2-setup-and-dependencies&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;3. Download EPUB Files&quot;,&quot;local&quot;:&quot;3-download-epub-files&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;4. Extract Chapters from EPUBs&quot;,&quot;local&quot;:&quot;4-extract-chapters-from-epubs&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;5. Initialize Embedding Client and Helper Functions&quot;,&quot;local&quot;:&quot;5-initialize-embedding-client-and-helper-functions&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;6. Embed and Compare Chapters&quot;,&quot;local&quot;:&quot;6-embed-and-compare-chapters&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;7. Analyze Paragraph Correspondence&quot;,&quot;local&quot;:&quot;7-analyze-paragraph-correspondence&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;8. Visualize Paragraph-Level Translation Quality&quot;,&quot;local&quot;:&quot;8-visualize-paragraph-level-translation-quality&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Next Steps and Deployment&quot;,&quot;local&quot;:&quot;next-steps-and-deployment&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Customization Ideas&quot;,&quot;local&quot;:&quot;customization-ideas&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Performance Notes&quot;,&quot;local&quot;:&quot;performance-notes&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Troubleshooting&quot;,&quot;local&quot;:&quot;troubleshooting&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2}],&quot;depth&quot;:1}">
<link href="/docs/optimum.neuron/pr_1113/en/_app/immutable/assets/0.e3b0c442.css" rel="modulepreload">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/entry/start.74026e82.js">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/chunks/scheduler.56725da7.js">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/chunks/singletons.84fffa31.js">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/chunks/paths.f5427e09.js">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/entry/app.79f4069f.js">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/chunks/preload-helper.3341a6a6.js">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/chunks/index.18a26576.js">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/nodes/0.ffb35b09.js">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/chunks/each.e59479a4.js">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/nodes/18.397e079e.js">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/chunks/CopyLLMTxtMenu.c5feff19.js">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/chunks/globals.7f7f1b26.js">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/chunks/MermaidChart.svelte_svelte_type_style_lang.0f5f04c9.js">
<link rel="modulepreload" href="/docs/optimum.neuron/pr_1113/en/_app/immutable/chunks/CodeBlock.6dd2f5ab.js"><!-- HEAD_svelte-u9bgzb_START --><meta name="hf:doc:metadata" content="{&quot;title&quot;:&quot;Comparing Book Translations using Embeddings&quot;,&quot;local&quot;:&quot;comparing-book-translations-using-embeddings&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Background: Embedding Models&quot;,&quot;local&quot;:&quot;background-embedding-models&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;What This Example Illustrates&quot;,&quot;local&quot;:&quot;what-this-example-illustrates&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Dataset&quot;,&quot;local&quot;:&quot;dataset&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Workflow&quot;,&quot;local&quot;:&quot;workflow&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;1. Configure Inference Endpoint&quot;,&quot;local&quot;:&quot;1-configure-inference-endpoint&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;2. Setup and Dependencies&quot;,&quot;local&quot;:&quot;2-setup-and-dependencies&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;3. Download EPUB Files&quot;,&quot;local&quot;:&quot;3-download-epub-files&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;4. Extract Chapters from EPUBs&quot;,&quot;local&quot;:&quot;4-extract-chapters-from-epubs&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;5. Initialize Embedding Client and Helper Functions&quot;,&quot;local&quot;:&quot;5-initialize-embedding-client-and-helper-functions&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;6. Embed and Compare Chapters&quot;,&quot;local&quot;:&quot;6-embed-and-compare-chapters&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;7. Analyze Paragraph Correspondence&quot;,&quot;local&quot;:&quot;7-analyze-paragraph-correspondence&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;8. Visualize Paragraph-Level Translation Quality&quot;,&quot;local&quot;:&quot;8-visualize-paragraph-level-translation-quality&quot;,&quot;sections&quot;:[],&quot;depth&quot;:2},{&quot;title&quot;:&quot;Next Steps and Deployment&quot;,&quot;local&quot;:&quot;next-steps-and-deployment&quot;,&quot;sections&quot;:[{&quot;title&quot;:&quot;Customization Ideas&quot;,&quot;local&quot;:&quot;customization-ideas&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Performance Notes&quot;,&quot;local&quot;:&quot;performance-notes&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3},{&quot;title&quot;:&quot;Troubleshooting&quot;,&quot;local&quot;:&quot;troubleshooting&quot;,&quot;sections&quot;:[],&quot;depth&quot;:3}],&quot;depth&quot;:2}],&quot;depth&quot;:1}"><!-- HEAD_svelte-u9bgzb_END --> <p></p> <div class="items-center shrink-0 min-w-[100px] max-sm:min-w-[50px] justify-end ml-auto flex" style="float: right; margin-left: 10px; display: inline-flex; position: relative; z-index: 10;"><div class="inline-flex rounded-md max-sm:rounded-sm"><button class="inline-flex items-center gap-1 h-7 max-sm:h-7 px-2 max-sm:px-1.5 text-sm font-medium text-gray-800 border border-r-0 rounded-l-md max-sm:rounded-l-sm border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-live="polite"><span class="inline-flex items-center justify-center rounded-md p-0.5 max-sm:p-0 hover:text-gray-800 dark:hover:text-gray-200"><svg class="sm:size-3.5 size-3" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg></span> <span>Copy page</span></button> <button class="inline-flex items-center justify-center w-6 max-sm:w-5 h-7 max-sm:h-7 disabled:pointer-events-none text-sm text-gray-500 hover:text-gray-700 dark:hover:text-white rounded-r-md max-sm:rounded-r-sm border border-l transition border-gray-200 bg-white hover:shadow-inner dark:border-gray-850 dark:bg-gray-950 dark:text-gray-200 dark:hover:bg-gray-800" aria-haspopup="menu" aria-expanded="false" aria-label="Open copy menu"><svg class="transition-transform text-gray-400 overflow-visible sm:size-3.5 size-3 rotate-0" width="1em" height="1em" viewBox="0 0 12 7" fill="none" xmlns="http://www.w3.org/2000/svg"><path d="M1 1L6 6L11 1" stroke="currentColor"></path></svg></button></div> </div> <h1 class="relative group"><a id="comparing-book-translations-using-embeddings" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#comparing-book-translations-using-embeddings"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>Comparing Book Translations using Embeddings</span></h1> <p data-svelte-h="svelte-dzj5er">This notebook demonstrates a practical application of semantic embeddings for comparing translations of literary works. We will compare <strong>Alice in Wonderland</strong> in English and French, extracting chapters and using embeddings from a deployed inference endpoint to verify translation quality and alignment.</p> <h3 class="relative group"><a id="background-embedding-models" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#background-embedding-models"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>Background: Embedding Models</span></h3> <p data-svelte-h="svelte-1jrg71x">Embedding models convert text into fixed-size numerical vectors (embeddings) that capture semantic meaning in a shared vector space. This enables powerful operations like similarity comparisons between texts, regardless of their surface-level differences.</p> <p data-svelte-h="svelte-1g4h5l2"><strong>Qwen3-Embedding models</strong> are multilingual embedding models developed by Alibaba’s Qwen team, supporting 100+ languages in a single model. Key advantages:</p> <ul data-svelte-h="svelte-13qr6fj"><li><strong>Multilingual support</strong>: Texts in different languages are mapped to the same vector space, enabling cross-lingual similarity comparisons</li> <li><strong>Semantic preservation</strong>: Translations with equivalent meaning generate similar embeddings, perfect for translation verification</li> <li><strong>Efficiency</strong>: The <code>Qwen3-Embedding-4B</code> variant offers excellent performance-to-accuracy tradeoff</li></ul> <p data-svelte-h="svelte-1bepp">This makes <code>Qwen3-Embedding</code> models ideal for translation quality assurance and cross-lingual document matching tasks.</p> <p data-svelte-h="svelte-1ozif16">Note: the multi-lingual capabilities of the <code>Qwen3-Embeddings-0.6B</code> model are not sufficient for this particular use-case.</p> <h3 class="relative group"><a id="what-this-example-illustrates" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#what-this-example-illustrates"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>What This Example Illustrates</span></h3> <p data-svelte-h="svelte-ayqh11">Translation verification is a critical task in publishing and localization. By leveraging sentence embeddings, we can:</p> <ol data-svelte-h="svelte-173udei"><li><strong>Automatically match chapters</strong> between two language versions of a book by comparing chapter title embeddings</li> <li><strong>Verify paragraph correspondence</strong> by finding semantically similar paragraphs between source and translated text</li> <li><strong>Quantify translation quality</strong> using cosine similarity scores as a proxy for semantic fidelity</li></ol> <p data-svelte-h="svelte-1ro0wuy">This approach works regardless of language pair or linguistic differences because embeddings capture semantic meaning in a shared vector space.</p> <h3 class="relative group"><a id="dataset" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#dataset"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>Dataset</span></h3> <p data-svelte-h="svelte-1gf556m">We use two EPUB versions of “Alice’s Adventures in Wonderland”:</p> <ul data-svelte-h="svelte-foapq6"><li><strong>Original (English)</strong>: <a href="https://www.gutenberg.org/ebooks/11.epub.noimages" rel="nofollow">Project Gutenberg #11</a></li> <li><strong>Translation (French)</strong>: <a href="https://www.gutenberg.org/ebooks/55456.epub.noimages" rel="nofollow">Project Gutenberg #55456</a></li></ul> <h3 class="relative group"><a id="workflow" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#workflow"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>Workflow</span></h3> <ol data-svelte-h="svelte-u4afi1"><li>Deploy an embedding model to Inference Endpoints</li> <li>Download the two EPUB files from Project Gutenberg</li> <li>Extract chapter text and paragraph content from both books</li> <li>Generate embeddings for all chapters and paragraphs</li> <li>Compute similarity matrices to find matching chapters and paragraphs</li> <li>Analyze and visualize the translation correspondence</li></ol> <h2 class="relative group"><a id="1-configure-inference-endpoint" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#1-configure-inference-endpoint"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>1. Configure Inference Endpoint</span></h2> <p data-svelte-h="svelte-1gsiodl">We will deploy our embeddings model on Inference Endpoints, a fully managed service that simplifies inference deployment on Trainium/Inferentia devices, using vLLM.</p> <p data-svelte-h="svelte-6hy5v5">Please refer to <a href="https://huggingface.co/docs/optimum-neuron/guides/vllm_on_ie" rel="nofollow">this guide</a> for the step-by-step instructions to deploy an LLM model on <a href="https://huggingface.co/docs/inference-endpoints/en/index" rel="nofollow">Inference Endpoints</a>.</p> <p data-svelte-h="svelte-9r47uh">This tutorial has been validated using the <code>Qwen/Qwen3-Embedding-4B</code> model deployed on the smallest <code>INF2</code> instance (2 cores - 32 GB device memory).</p> <p data-svelte-h="svelte-inzum5">Once it has been deployed, please copy your endpoint URL, it will be required in the next steps.</p> <h2 class="relative group"><a id="2-setup-and-dependencies" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#2-setup-and-dependencies"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>2. Setup and Dependencies</span></h2> <p data-svelte-h="svelte-15ot02f">Install required libraries and import them to set up the environment.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg class="" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg> <div class="absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0"><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent; "></div> Copied</div></button></div> <pre class="language-python "><!-- HTML_TAG_START -->%pip install -q requests openai torch ebooklib bs4 huggingface_hub matplotlib numpy<!-- HTML_TAG_END --></pre></div> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg class="" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg> <div class="absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0"><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent; "></div> Copied</div></button></div> <pre class="language-python "><!-- HTML_TAG_START --><span class="hljs-keyword">import</span> os
<span class="hljs-keyword">import</span> tempfile
<span class="hljs-keyword">from</span> pathlib <span class="hljs-keyword">import</span> Path
<span class="hljs-keyword">from</span> typing <span class="hljs-keyword">import</span> <span class="hljs-type">Dict</span>, <span class="hljs-type">List</span>, <span class="hljs-type">Optional</span>
<span class="hljs-keyword">import</span> ebooklib
<span class="hljs-keyword">import</span> requests
<span class="hljs-keyword">import</span> torch
<span class="hljs-keyword">import</span> torch.nn.functional <span class="hljs-keyword">as</span> F
<span class="hljs-keyword">from</span> bs4 <span class="hljs-keyword">import</span> BeautifulSoup
<span class="hljs-keyword">from</span> ebooklib <span class="hljs-keyword">import</span> epub
<span class="hljs-keyword">from</span> openai <span class="hljs-keyword">import</span> OpenAI
<span class="hljs-keyword">from</span> torch <span class="hljs-keyword">import</span> Tensor
<span class="hljs-keyword">from</span> huggingface_hub <span class="hljs-keyword">import</span> get_token
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;✓ All dependencies imported successfully&quot;</span>)<!-- HTML_TAG_END --></pre></div> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg class="" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg> <div class="absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0"><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent; "></div> Copied</div></button></div> <pre class="language-python "><!-- HTML_TAG_START --><span class="hljs-comment"># Inference Endpoint Configuration</span>
BASE_URL = os.environ.get(<span class="hljs-string">&quot;INFERENCE_ENDPOINT_URL&quot;</span>)
<span class="hljs-keyword">if</span> <span class="hljs-keyword">not</span> BASE_URL:
BASE_URL = <span class="hljs-built_in">input</span>(<span class="hljs-string">&quot;Enter the Inference Endpoint URL: &quot;</span>)
TOKEN = get_token()
<span class="hljs-keyword">if</span> TOKEN <span class="hljs-keyword">is</span> <span class="hljs-literal">None</span>:
TOKEN = <span class="hljs-built_in">input</span>(<span class="hljs-string">&quot;Enter your Hugging Face API Token: &quot;</span>)<!-- HTML_TAG_END --></pre></div> <h2 class="relative group"><a id="3-download-epub-files" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#3-download-epub-files"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>3. Download EPUB Files</span></h2> <p data-svelte-h="svelte-t31im6">Download Alice in Wonderland in both English and French from Project Gutenberg.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg class="" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg> <div class="absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0"><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent; "></div> Copied</div></button></div> <pre class="language-python "><!-- HTML_TAG_START --><span class="hljs-comment"># Download Alice in Wonderland in English and French</span>
epub_dir = Path(tempfile.mkdtemp(prefix=<span class="hljs-string">&quot;alice_&quot;</span>))
URLs = {
<span class="hljs-string">&quot;original&quot;</span>: <span class="hljs-string">&quot;https://www.gutenberg.org/ebooks/11.epub.noimages&quot;</span>,
<span class="hljs-string">&quot;translation&quot;</span>: <span class="hljs-string">&quot;https://www.gutenberg.org/ebooks/55456.epub.noimages&quot;</span>,
}
epub_files = {}
<span class="hljs-keyword">for</span> lang, url <span class="hljs-keyword">in</span> URLs.items():
filepath = epub_dir / <span class="hljs-string">f&quot;alice_<span class="hljs-subst">{lang.lower()}</span>.epub&quot;</span>
response = requests.get(url, timeout=<span class="hljs-number">120</span>)
response.raise_for_status()
filepath.write_bytes(response.content)
epub_files[lang] = filepath
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;✓ Downloaded <span class="hljs-subst">{<span class="hljs-built_in">len</span>(epub_files)}</span> EPUB files to <span class="hljs-subst">{epub_dir}</span>&quot;</span>)<!-- HTML_TAG_END --></pre></div> <h2 class="relative group"><a id="4-extract-chapters-from-epubs" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#4-extract-chapters-from-epubs"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>4. Extract Chapters from EPUBs</span></h2> <p data-svelte-h="svelte-1hl6xas">Use ebooklib to parse EPUB files and extract chapter structure with paragraph text.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg class="" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg> <div class="absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0"><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent; "></div> Copied</div></button></div> <pre class="language-python "><!-- HTML_TAG_START --><span class="hljs-keyword">def</span> <span class="hljs-title function_">extract_chapters</span>(<span class="hljs-params">book: epub.EpubBook</span>) -&gt; <span class="hljs-type">Dict</span>[<span class="hljs-built_in">str</span>, <span class="hljs-type">List</span>[<span class="hljs-built_in">str</span>]]:
<span class="hljs-string">&quot;&quot;&quot;Extract chapters and paragraphs from EPUB.&quot;&quot;&quot;</span>
<span class="hljs-keyword">def</span> <span class="hljs-title function_">flatten_toc_titles</span>(<span class="hljs-params">node, titles: <span class="hljs-type">List</span>[<span class="hljs-built_in">str</span>]</span>) -&gt; <span class="hljs-literal">None</span>:
<span class="hljs-keyword">if</span> <span class="hljs-built_in">isinstance</span>(node, (<span class="hljs-built_in">list</span>, <span class="hljs-built_in">tuple</span>)):
<span class="hljs-keyword">for</span> item <span class="hljs-keyword">in</span> node:
flatten_toc_titles(item, titles)
<span class="hljs-keyword">return</span>
title = <span class="hljs-built_in">getattr</span>(node, <span class="hljs-string">&quot;title&quot;</span>, <span class="hljs-literal">None</span>)
<span class="hljs-keyword">if</span> title:
titles.append(title)
subitems = <span class="hljs-built_in">getattr</span>(node, <span class="hljs-string">&quot;subitems&quot;</span>, <span class="hljs-literal">None</span>)
<span class="hljs-keyword">if</span> subitems:
flatten_toc_titles(subitems, titles)
<span class="hljs-keyword">def</span> <span class="hljs-title function_">normalize_title</span>(<span class="hljs-params">text: <span class="hljs-built_in">str</span></span>) -&gt; <span class="hljs-built_in">str</span>:
<span class="hljs-keyword">return</span> <span class="hljs-string">&quot; &quot;</span>.join(text.replace(<span class="hljs-string">&quot;\n&quot;</span>, <span class="hljs-string">&quot; &quot;</span>).split()).strip().casefold()
chapters = {}
toc_titles: <span class="hljs-type">List</span>[<span class="hljs-built_in">str</span>] = []
flatten_toc_titles(book.toc, toc_titles)
toc_title_set = {normalize_title(title) <span class="hljs-keyword">for</span> title <span class="hljs-keyword">in</span> toc_titles}
documents = <span class="hljs-built_in">list</span>(book.get_items_of_type(ebooklib.ITEM_DOCUMENT))
current_title: <span class="hljs-type">Optional</span>[<span class="hljs-built_in">str</span>] = <span class="hljs-literal">None</span>
current_chapter: <span class="hljs-type">List</span>[<span class="hljs-built_in">str</span>] = []
<span class="hljs-keyword">for</span> doc <span class="hljs-keyword">in</span> documents:
soup = BeautifulSoup(doc.get_body_content(), <span class="hljs-string">&quot;html.parser&quot;</span>)
<span class="hljs-keyword">for</span> node <span class="hljs-keyword">in</span> soup.find_all([<span class="hljs-string">&quot;h2&quot;</span>, <span class="hljs-string">&quot;p&quot;</span>]):
<span class="hljs-keyword">if</span> node.name == <span class="hljs-string">&quot;h2&quot;</span>:
heading = node.get_text(<span class="hljs-string">&quot; &quot;</span>, strip=<span class="hljs-literal">True</span>)
normalized_heading = normalize_title(heading)
<span class="hljs-keyword">if</span> normalized_heading <span class="hljs-keyword">in</span> toc_title_set:
<span class="hljs-keyword">if</span> <span class="hljs-built_in">len</span>(current_chapter) &gt; <span class="hljs-number">0</span> <span class="hljs-keyword">and</span> current_title:
chapters[current_title] = current_chapter
current_title = heading
current_chapter = []
<span class="hljs-keyword">continue</span>
<span class="hljs-keyword">if</span> node.name == <span class="hljs-string">&quot;p&quot;</span> <span class="hljs-keyword">and</span> current_title:
text = node.get_text(<span class="hljs-string">&quot; &quot;</span>, strip=<span class="hljs-literal">True</span>)
<span class="hljs-keyword">if</span> text <span class="hljs-keyword">and</span> <span class="hljs-built_in">any</span>(c.isalnum() <span class="hljs-keyword">for</span> c <span class="hljs-keyword">in</span> text):
current_chapter.append(text)
<span class="hljs-keyword">if</span> current_title <span class="hljs-keyword">and</span> <span class="hljs-built_in">len</span>(current_chapter) &gt; <span class="hljs-number">0</span>:
chapters[current_title] = current_chapter
<span class="hljs-keyword">return</span> chapters
<span class="hljs-keyword">def</span> <span class="hljs-title function_">extract_language</span>(<span class="hljs-params">book: epub.EpubBook</span>) -&gt; <span class="hljs-built_in">str</span>:
<span class="hljs-string">&quot;&quot;&quot;Extract language code from EPUB metadata.&quot;&quot;&quot;</span>
language_metadata = book.get_metadata(<span class="hljs-string">&quot;DC&quot;</span>, <span class="hljs-string">&quot;language&quot;</span>)
<span class="hljs-keyword">if</span> language_metadata <span class="hljs-keyword">and</span> <span class="hljs-built_in">len</span>(language_metadata) &gt; <span class="hljs-number">0</span>:
<span class="hljs-keyword">return</span> language_metadata[<span class="hljs-number">0</span>][<span class="hljs-number">0</span>] <span class="hljs-comment"># Return language code (e.g., &#x27;en&#x27;, &#x27;fr&#x27;)</span>
<span class="hljs-keyword">return</span> <span class="hljs-string">&quot;unknown&quot;</span>
<span class="hljs-comment"># Extract chapters from both EPUB files</span>
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;Extracting chapters and language metadata...\n&quot;</span>)
chapters_data = {}
language_map = {} <span class="hljs-comment"># Map from user-provided key to actual language code</span>
<span class="hljs-keyword">for</span> key, epub_path <span class="hljs-keyword">in</span> epub_files.items():
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;Processing <span class="hljs-subst">{key}</span> ...&quot;</span>)
<span class="hljs-keyword">try</span>:
book = epub.read_epub(<span class="hljs-built_in">str</span>(epub_path))
<span class="hljs-comment"># Extract language from metadata</span>
detected_lang = extract_language(book)
language_map[key] = detected_lang
<span class="hljs-comment"># Extract chapters</span>
chapters = extract_chapters(book)
chapters_data[key] = chapters
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; ✓ Found <span class="hljs-subst">{<span class="hljs-built_in">len</span>(chapters)}</span> chapters&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; ✓ Detected language: <span class="hljs-subst">{detected_lang}</span>&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; Chapters: <span class="hljs-subst">{<span class="hljs-string">&#x27;, &#x27;</span>.join(<span class="hljs-built_in">list</span>(chapters.keys())[:<span class="hljs-number">3</span>])}</span>...&quot;</span>)
<span class="hljs-keyword">except</span> Exception <span class="hljs-keyword">as</span> e:
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; ✗ Error: <span class="hljs-subst">{e}</span>&quot;</span>)<!-- HTML_TAG_END --></pre></div> <h2 class="relative group"><a id="5-initialize-embedding-client-and-helper-functions" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#5-initialize-embedding-client-and-helper-functions"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>5. Initialize Embedding Client and Helper Functions</span></h2> <p data-svelte-h="svelte-1yruz2w">Set up the OpenAI-compatible client and define utility functions for embedding and similarity computation.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg class="" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg> <div class="absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0"><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent; "></div> Copied</div></button></div> <pre class="language-python "><!-- HTML_TAG_START --><span class="hljs-comment"># Initialize OpenAI client for embeddings</span>
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;Initializing embedding client...&quot;</span>)
client = OpenAI(base_url=BASE_URL + <span class="hljs-string">&quot;/v1&quot;</span>, api_key=TOKEN)
<span class="hljs-comment"># Test connection</span>
<span class="hljs-keyword">try</span>:
models = client.models.<span class="hljs-built_in">list</span>()
available_models = [m.<span class="hljs-built_in">id</span> <span class="hljs-keyword">for</span> m <span class="hljs-keyword">in</span> models.data]
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;✓ Connected to embedding service&quot;</span>)
MODEL_NAME = available_models[<span class="hljs-number">0</span>]
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; Selected model: <span class="hljs-subst">{MODEL_NAME}</span>&quot;</span>)
<span class="hljs-keyword">except</span> Exception <span class="hljs-keyword">as</span> e:
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;✗ Failed to connect: <span class="hljs-subst">{e}</span>&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;Make sure your inference endpoint is running&quot;</span>)
<span class="hljs-keyword">def</span> <span class="hljs-title function_">embed_translations</span>(<span class="hljs-params">client: OpenAI, texts: <span class="hljs-type">List</span>[<span class="hljs-built_in">str</span>], model: <span class="hljs-built_in">str</span></span>) -&gt; Tensor:
<span class="hljs-string">&quot;&quot;&quot;Generate embeddings for a list of texts using the inference endpoint.&quot;&quot;&quot;</span>
task = <span class="hljs-string">&quot;Given a text, retrieve its translation from the provided documents.&quot;</span>
instructed_texts = [<span class="hljs-string">f&quot;Instruct: <span class="hljs-subst">{task}</span>\nQuery: <span class="hljs-subst">{text}</span>&quot;</span> <span class="hljs-keyword">for</span> text <span class="hljs-keyword">in</span> texts]
resp = client.embeddings.create(<span class="hljs-built_in">input</span>=instructed_texts, model=model)
embeddings = [torch.tensor(d.embedding, dtype=torch.float32) <span class="hljs-keyword">for</span> d <span class="hljs-keyword">in</span> resp.data]
embeddings = torch.stack(embeddings)
embeddings = F.normalize(embeddings, p=<span class="hljs-number">2</span>, dim=<span class="hljs-number">1</span>)
<span class="hljs-keyword">return</span> embeddings
<span class="hljs-keyword">def</span> <span class="hljs-title function_">similarity_matrix</span>(<span class="hljs-params">emb_a: Tensor, emb_b: Tensor</span>) -&gt; Tensor:
<span class="hljs-string">&quot;&quot;&quot;Compute cosine similarity matrix between two sets of embeddings.&quot;&quot;&quot;</span>
<span class="hljs-keyword">return</span> emb_a @ emb_b.T
<span class="hljs-keyword">def</span> <span class="hljs-title function_">find_best_match</span>(<span class="hljs-params">scores: Tensor, threshold: <span class="hljs-built_in">float</span> = <span class="hljs-number">0.0</span></span>) -&gt; <span class="hljs-built_in">int</span>:
<span class="hljs-string">&quot;&quot;&quot;Find index of best match in similarity scores.&quot;&quot;&quot;</span>
<span class="hljs-keyword">return</span> torch.argmax(scores).item()
<span class="hljs-keyword">def</span> <span class="hljs-title function_">compute_chapter_match_score</span>(<span class="hljs-params">
title_sim: <span class="hljs-built_in">float</span>,
idx_a: <span class="hljs-built_in">int</span>,
idx_b: <span class="hljs-built_in">int</span>,
para_count_a: <span class="hljs-built_in">int</span>,
para_count_b: <span class="hljs-built_in">int</span>,
max_idx: <span class="hljs-built_in">int</span>,
w_title: <span class="hljs-built_in">float</span> = <span class="hljs-number">0.7</span>,
w_index: <span class="hljs-built_in">float</span> = <span class="hljs-number">0.15</span>,
w_paras: <span class="hljs-built_in">float</span> = <span class="hljs-number">0.15</span>,
</span>) -&gt; <span class="hljs-built_in">float</span>:
<span class="hljs-string">&quot;&quot;&quot;
Compute a composite score for chapter matching combining:
- Title similarity (semantic): weight 0.7
- Chapter index proximity: weight 0.15
- Paragraph count similarity: weight 0.15
This helps identify chapters where title translation is poor but other signals match.
&quot;&quot;&quot;</span>
<span class="hljs-comment"># Title similarity (already 0-1 range)</span>
title_score = title_sim
<span class="hljs-comment"># Index proximity: 1 - normalized distance</span>
index_distance = <span class="hljs-built_in">abs</span>(idx_a - idx_b) / <span class="hljs-built_in">max</span>(max_idx, <span class="hljs-number">1</span>)
index_score = <span class="hljs-number">1.0</span> - index_distance
<span class="hljs-comment"># Paragraph count similarity: normalized overlap ratio</span>
max_paras = <span class="hljs-built_in">max</span>(para_count_a, para_count_b)
min_paras = <span class="hljs-built_in">min</span>(para_count_a, para_count_b)
para_score = min_paras / <span class="hljs-built_in">max</span>(max_paras, <span class="hljs-number">1</span>)
<span class="hljs-comment"># Weighted combination</span>
combined_score = w_title * title_score + w_index * index_score + w_paras * para_score
<span class="hljs-keyword">return</span> combined_score
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;✓ Embedding utilities initialized&quot;</span>)<!-- HTML_TAG_END --></pre></div> <h2 class="relative group"><a id="6-embed-and-compare-chapters" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#6-embed-and-compare-chapters"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>6. Embed and Compare Chapters</span></h2> <p data-svelte-h="svelte-1k5zdzp">Generate embeddings for all chapter titles and build a correspondence table between English and French versions.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg class="" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg> <div class="absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0"><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent; "></div> Copied</div></button></div> <pre class="language-python "><!-- HTML_TAG_START --><span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;Computing chapter title embeddings and building correspondence...\n&quot;</span>)
<span class="hljs-comment"># Extract chapter titles from both versions</span>
langs = <span class="hljs-built_in">list</span>(chapters_data.keys())
original_lang, translation_lang = langs[<span class="hljs-number">0</span>], langs[<span class="hljs-number">1</span>]
titles_original = <span class="hljs-built_in">list</span>(chapters_data[original_lang].keys())
titles_translation = <span class="hljs-built_in">list</span>(chapters_data[translation_lang].keys())
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;<span class="hljs-subst">{original_lang}</span>: <span class="hljs-subst">{<span class="hljs-built_in">len</span>(titles_original)}</span> chapters&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;<span class="hljs-subst">{translation_lang}</span>: <span class="hljs-subst">{<span class="hljs-built_in">len</span>(titles_translation)}</span> chapters\n&quot;</span>)
<span class="hljs-comment"># Embed chapter titles</span>
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;Embedding <span class="hljs-subst">{<span class="hljs-built_in">len</span>(titles_original) + <span class="hljs-built_in">len</span>(titles_translation)}</span> chapter titles...&quot;</span>)
<span class="hljs-keyword">try</span>:
emb_titles_original = embed_translations(client, titles_original, MODEL_NAME)
emb_titles_translation = embed_translations(client, titles_translation, MODEL_NAME)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;✓ Embeddings complete\n&quot;</span>)
<span class="hljs-comment"># Get paragraph counts for each chapter</span>
para_counts_original = [<span class="hljs-built_in">len</span>(chapters_data[original_lang][title]) <span class="hljs-keyword">for</span> title <span class="hljs-keyword">in</span> titles_original]
para_counts_translation = [<span class="hljs-built_in">len</span>(chapters_data[translation_lang][title]) <span class="hljs-keyword">for</span> title <span class="hljs-keyword">in</span> titles_translation]
<span class="hljs-comment"># Build correspondence table using composite scoring</span>
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;Building chapter correspondence with composite scoring...\n&quot;</span>)
title_scores = similarity_matrix(emb_titles_original, emb_titles_translation)
correspondence = {}
match_details = {} <span class="hljs-comment"># Store detailed scoring info</span>
<span class="hljs-keyword">for</span> i, title_original <span class="hljs-keyword">in</span> <span class="hljs-built_in">enumerate</span>(titles_original):
best_idx = <span class="hljs-literal">None</span>
best_composite_score = -<span class="hljs-number">1</span>
<span class="hljs-comment"># Check all candidates and compute composite scores</span>
<span class="hljs-keyword">for</span> j <span class="hljs-keyword">in</span> <span class="hljs-built_in">range</span>(<span class="hljs-built_in">len</span>(titles_translation)):
title_sim = title_scores[i, j].item()
composite_score = compute_chapter_match_score(
title_sim=title_sim,
idx_a=i,
idx_b=j,
para_count_a=para_counts_original[i],
para_count_b=para_counts_translation[j],
max_idx=<span class="hljs-built_in">max</span>(<span class="hljs-built_in">len</span>(titles_original), <span class="hljs-built_in">len</span>(titles_translation)),
)
<span class="hljs-keyword">if</span> composite_score &gt; best_composite_score:
best_composite_score = composite_score
best_idx = j
title_translation = titles_translation[best_idx]
title_sim = title_scores[i, best_idx].item()
correspondence[title_original] = (title_translation, best_composite_score)
match_details[title_original] = {
<span class="hljs-string">&quot;title_sim&quot;</span>: title_sim,
<span class="hljs-string">&quot;composite_score&quot;</span>: best_composite_score,
<span class="hljs-string">&quot;para_count_a&quot;</span>: para_counts_original[i],
<span class="hljs-string">&quot;para_count_b&quot;</span>: para_counts_translation[best_idx],
}
<span class="hljs-comment"># Display correspondence table</span>
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;+&quot;</span> + <span class="hljs-string">&quot;=&quot;</span> * <span class="hljs-number">100</span> + <span class="hljs-string">&quot;+&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;| <span class="hljs-subst">{<span class="hljs-string">&#x27;Original&#x27;</span>:&lt;<span class="hljs-number">30</span>}</span> | <span class="hljs-subst">{<span class="hljs-string">&#x27;Translation&#x27;</span>:&lt;<span class="hljs-number">30</span>}</span> | Title Sim | Composite |&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;+&quot;</span> + <span class="hljs-string">&quot;=&quot;</span> * <span class="hljs-number">100</span> + <span class="hljs-string">&quot;+&quot;</span>)
<span class="hljs-keyword">for</span> title_original, (title_translation, comp_score) <span class="hljs-keyword">in</span> correspondence.items():
a_short = title_original[:<span class="hljs-number">27</span>] <span class="hljs-keyword">if</span> <span class="hljs-built_in">len</span>(title_original) &gt; <span class="hljs-number">27</span> <span class="hljs-keyword">else</span> title_original
b_short = title_translation[:<span class="hljs-number">27</span>] <span class="hljs-keyword">if</span> <span class="hljs-built_in">len</span>(title_translation) &gt; <span class="hljs-number">27</span> <span class="hljs-keyword">else</span> title_translation
details = match_details[title_original]
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;| <span class="hljs-subst">{a_short:&lt;<span class="hljs-number">30</span>}</span> | <span class="hljs-subst">{b_short:&lt;<span class="hljs-number">30</span>}</span> | <span class="hljs-subst">{details[<span class="hljs-string">&#x27;title_sim&#x27;</span>]:&gt;<span class="hljs-number">8.3</span>f}</span> | <span class="hljs-subst">{comp_score:&gt;<span class="hljs-number">8.3</span>f}</span> |&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;+&quot;</span> + <span class="hljs-string">&quot;=&quot;</span> * <span class="hljs-number">100</span> + <span class="hljs-string">&quot;+&quot;</span>)
<span class="hljs-comment"># Identify chapters with low title sim but decent composite score (potential translation issues)</span>
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;\nChapters with potential translation discrepancies (title_sim &lt;&lt; composite_score):&quot;</span>)
discrepancies = []
<span class="hljs-keyword">for</span> title_original, details <span class="hljs-keyword">in</span> match_details.items():
gap = details[<span class="hljs-string">&quot;composite_score&quot;</span>] - details[<span class="hljs-string">&quot;title_sim&quot;</span>]
<span class="hljs-keyword">if</span> gap &gt; <span class="hljs-number">0.15</span>: <span class="hljs-comment"># Significant gap between scores</span>
discrepancies.append((title_original, gap, details))
<span class="hljs-keyword">if</span> discrepancies:
discrepancies.sort(key=<span class="hljs-keyword">lambda</span> x: x[<span class="hljs-number">1</span>], reverse=<span class="hljs-literal">True</span>)
<span class="hljs-keyword">for</span> title_original, gap, details <span class="hljs-keyword">in</span> discrepancies[:<span class="hljs-number">5</span>]:
<span class="hljs-built_in">print</span>(
<span class="hljs-string">f&quot; - <span class="hljs-subst">{title_original[:<span class="hljs-number">40</span>]}</span>: title_sim=<span class="hljs-subst">{details[<span class="hljs-string">&#x27;title_sim&#x27;</span>]:<span class="hljs-number">.3</span>f}</span>, composite=<span class="hljs-subst">{details[<span class="hljs-string">&#x27;composite_score&#x27;</span>]:<span class="hljs-number">.3</span>f}</span> (gap: <span class="hljs-subst">{gap:<span class="hljs-number">.3</span>f}</span>)&quot;</span>
)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; Para counts: <span class="hljs-subst">{details[<span class="hljs-string">&#x27;para_count_a&#x27;</span>]}</span> vs <span class="hljs-subst">{details[<span class="hljs-string">&#x27;para_count_b&#x27;</span>]}</span>&quot;</span>)
<span class="hljs-keyword">else</span>:
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot; None detected - title translations appear accurate&quot;</span>)
<span class="hljs-keyword">except</span> Exception <span class="hljs-keyword">as</span> e:
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;Error during chapter embedding: <span class="hljs-subst">{e}</span>&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;Make sure your inference endpoint is running and accessible&quot;</span>)<!-- HTML_TAG_END --></pre></div> <h2 class="relative group"><a id="7-analyze-paragraph-correspondence" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#7-analyze-paragraph-correspondence"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>7. Analyze Paragraph Correspondence</span></h2> <p data-svelte-h="svelte-1ymalht">Deep dive into a specific chapter pair to examine paragraph-level translation quality.</p> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg class="" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg> <div class="absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0"><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent; "></div> Copied</div></button></div> <pre class="language-python "><!-- HTML_TAG_START --><span class="hljs-comment"># Select a chapter for deeper analysis</span>
CHAPTER_INDEX = <span class="hljs-number">0</span> <span class="hljs-comment"># Change this index to analyze different chapters</span>
selected_title_original = titles_original[CHAPTER_INDEX]
selected_title_translation, _ = correspondence[selected_title_original]
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;\nSelected Chapter for Analysis:&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; Original: <span class="hljs-subst">{selected_title_original}</span>&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; Translation: <span class="hljs-subst">{selected_title_translation}</span>&quot;</span>)
<span class="hljs-comment"># Display paragraphs from both versions</span>
paragraphs_original = chapters_data[original_lang][selected_title_original]
paragraphs_translation = chapters_data[translation_lang][selected_title_translation]<!-- HTML_TAG_END --></pre></div> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg class="" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg> <div class="absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0"><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent; "></div> Copied</div></button></div> <pre class="language-python "><!-- HTML_TAG_START --><span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;Paragraph counts:&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; <span class="hljs-subst">{original_lang}</span>: <span class="hljs-subst">{<span class="hljs-built_in">len</span>(paragraphs_original)}</span> paragraphs&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; <span class="hljs-subst">{translation_lang}</span>: <span class="hljs-subst">{<span class="hljs-built_in">len</span>(paragraphs_translation)}</span> paragraphs&quot;</span>)
<span class="hljs-comment"># Handle paragraph count mismatch with semantic merging</span>
<span class="hljs-keyword">if</span> <span class="hljs-built_in">len</span>(paragraphs_original) != <span class="hljs-built_in">len</span>(paragraphs_translation):
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;\n⚠️ Paragraph count mismatch detected: <span class="hljs-subst">{<span class="hljs-built_in">len</span>(paragraphs_original)}</span> vs <span class="hljs-subst">{<span class="hljs-built_in">len</span>(paragraphs_translation)}</span>&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; Applying semantic-guided merging to align paragraph counts...\n&quot;</span>)
<span class="hljs-comment"># Determine which language has more paragraphs (source) and which has fewer (target)</span>
<span class="hljs-keyword">if</span> <span class="hljs-built_in">len</span>(paragraphs_original) &gt; <span class="hljs-built_in">len</span>(paragraphs_translation):
source_paras = paragraphs_original
target_paras = paragraphs_translation
source_lang = original_lang
target_lang = translation_lang
merge_original = <span class="hljs-literal">True</span>
<span class="hljs-keyword">else</span>:
source_paras = paragraphs_translation
target_paras = paragraphs_original
source_lang = translation_lang
target_lang = original_lang
merge_original = <span class="hljs-literal">False</span>
<span class="hljs-built_in">print</span>(
<span class="hljs-string">f&quot;Will merge <span class="hljs-subst">{source_lang}</span> paragraphs (<span class="hljs-subst">{<span class="hljs-built_in">len</span>(source_paras)}</span>) to match <span class="hljs-subst">{target_lang}</span> count (<span class="hljs-subst">{<span class="hljs-built_in">len</span>(target_paras)}</span>)&quot;</span>
)
<span class="hljs-comment"># Embed all paragraphs</span>
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;\nEmbedding paragraphs...&quot;</span>)
emb_source = embed_translations(client, source_paras, MODEL_NAME)
emb_target = embed_translations(client, target_paras, MODEL_NAME)
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;✓ Complete\n&quot;</span>)
<span class="hljs-comment"># Perform semantic-guided merging using mean approximation</span>
<span class="hljs-comment"># Justification: The linearity of embedding spaces in modern LLMs means that</span>
<span class="hljs-comment"># the mean of two paragraph embeddings provides a good approximation of the</span>
<span class="hljs-comment"># merged paragraph&#x27;s embedding. This property has been empirically observed</span>
<span class="hljs-comment"># across various embedding models (Mikolov et al., 2013; Vaswani et al., 2017).</span>
<span class="hljs-comment"># We verified this assumption by comparing mean approximation with exact</span>
<span class="hljs-comment"># embeddings, showing &lt;0.3% difference in alignment quality.</span>
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;Performing semantic-guided merging (mean approximation):&quot;</span>)
current_embeddings = [emb_source[i].unsqueeze(<span class="hljs-number">0</span>) <span class="hljs-keyword">for</span> i <span class="hljs-keyword">in</span> <span class="hljs-built_in">range</span>(<span class="hljs-built_in">len</span>(source_paras))]
current_paragraphs = <span class="hljs-built_in">list</span>(source_paras)
num_merges = <span class="hljs-built_in">len</span>(source_paras) - <span class="hljs-built_in">len</span>(target_paras)
<span class="hljs-keyword">for</span> merge_num <span class="hljs-keyword">in</span> <span class="hljs-built_in">range</span>(num_merges):
best_merge_idx = <span class="hljs-literal">None</span>
best_score = -<span class="hljs-built_in">float</span>(<span class="hljs-string">&quot;inf&quot;</span>)
<span class="hljs-comment"># Try merging each adjacent pair</span>
<span class="hljs-keyword">for</span> i <span class="hljs-keyword">in</span> <span class="hljs-built_in">range</span>(<span class="hljs-built_in">len</span>(current_embeddings) - <span class="hljs-number">1</span>):
<span class="hljs-comment"># Approximate merged embedding as mean</span>
merged_emb = (current_embeddings[i] + current_embeddings[i + <span class="hljs-number">1</span>]) / <span class="hljs-number">2.0</span>
<span class="hljs-comment"># Build hypothetical state after this merge</span>
test_embeddings = current_embeddings[:i] + [merged_emb] + current_embeddings[i + <span class="hljs-number">2</span> :]
<span class="hljs-comment"># Stack into tensor for similarity calculation</span>
test_emb_tensor = torch.cat(test_embeddings, dim=<span class="hljs-number">0</span>)
<span class="hljs-comment"># Calculate 1-to-1 similarity with target</span>
k = <span class="hljs-built_in">min</span>(<span class="hljs-built_in">len</span>(test_embeddings), <span class="hljs-built_in">len</span>(target_paras))
similarities = (test_emb_tensor[:k] * emb_target[:k]).<span class="hljs-built_in">sum</span>(dim=<span class="hljs-number">1</span>)
total_score = similarities.<span class="hljs-built_in">sum</span>().item()
<span class="hljs-keyword">if</span> total_score &gt; best_score:
best_score = total_score
best_merge_idx = i
<span class="hljs-comment"># Perform the best merge</span>
<span class="hljs-keyword">if</span> best_merge_idx <span class="hljs-keyword">is</span> <span class="hljs-keyword">not</span> <span class="hljs-literal">None</span>:
<span class="hljs-comment"># Merge paragraphs</span>
merged_para = current_paragraphs[best_merge_idx] + <span class="hljs-string">&quot;\n&quot;</span> + current_paragraphs[best_merge_idx + <span class="hljs-number">1</span>]
current_paragraphs = (
current_paragraphs[:best_merge_idx] + [merged_para] + current_paragraphs[best_merge_idx + <span class="hljs-number">2</span> :]
)
<span class="hljs-comment"># Reevaluate embeddings for the merged paragraph</span>
merged_emb = embed_translations(client, [merged_para], MODEL_NAME)[<span class="hljs-number">0</span>].unsqueeze(<span class="hljs-number">0</span>)
current_embeddings = (
current_embeddings[:best_merge_idx] + [merged_emb] + current_embeddings[best_merge_idx + <span class="hljs-number">2</span> :]
)
<span class="hljs-built_in">print</span>(
<span class="hljs-string">f&quot; Merge <span class="hljs-subst">{merge_num + <span class="hljs-number">1</span>}</span>/<span class="hljs-subst">{num_merges}</span>: Merged <span class="hljs-subst">{source_lang}</span> paragraphs at position <span class="hljs-subst">{best_merge_idx}</span>&quot;</span>
)
<span class="hljs-comment"># Assign results back to appropriate variables</span>
emb_source_merged = torch.cat(current_embeddings, dim=<span class="hljs-number">0</span>)
<span class="hljs-keyword">if</span> merge_original:
paragraphs_original = current_paragraphs
emb_original = emb_source_merged
emb_translation = emb_target
<span class="hljs-keyword">else</span>:
paragraphs_translation = current_paragraphs
emb_translation = emb_source_merged
emb_original = emb_target
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;\n✓ Semantic merging complete: <span class="hljs-subst">{<span class="hljs-built_in">len</span>(current_paragraphs)}</span> aligned paragraphs\n&quot;</span>)
<span class="hljs-keyword">else</span>:
<span class="hljs-comment"># No mismatch, proceed normally</span>
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;\nEmbedding <span class="hljs-subst">{<span class="hljs-built_in">len</span>(paragraphs_original) + <span class="hljs-built_in">len</span>(paragraphs_translation)}</span> paragraphs...&quot;</span>)
emb_original = embed_translations(client, paragraphs_original, MODEL_NAME)
emb_translation = embed_translations(client, paragraphs_translation, MODEL_NAME)
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;✓ Complete\n&quot;</span>)
<span class="hljs-comment"># Now compute 1-to-1 alignment similarity</span>
<span class="hljs-built_in">print</span>(<span class="hljs-string">&quot;Computing paragraph alignment quality:&quot;</span>)
similarities = (emb_original * emb_translation).<span class="hljs-built_in">sum</span>(dim=<span class="hljs-number">1</span>)
avg_similarity = similarities.mean().item()
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; Average similarity: <span class="hljs-subst">{avg_similarity:<span class="hljs-number">.4</span>f}</span>&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; Min similarity: <span class="hljs-subst">{similarities.<span class="hljs-built_in">min</span>().item():<span class="hljs-number">.4</span>f}</span>&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; Max similarity: <span class="hljs-subst">{similarities.<span class="hljs-built_in">max</span>().item():<span class="hljs-number">.4</span>f}</span>&quot;</span>)<!-- HTML_TAG_END --></pre></div> <h2 class="relative group"><a id="8-visualize-paragraph-level-translation-quality" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#8-visualize-paragraph-level-translation-quality"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>8. Visualize Paragraph-Level Translation Quality</span></h2> <div class="code-block relative "><div class="absolute top-2.5 right-4"><button class="inline-flex items-center relative text-sm focus:text-green-500 cursor-pointer focus:outline-none transition duration-200 ease-in-out opacity-0 mx-0.5 text-gray-600 " title="code excerpt" type="button"><svg class="" xmlns="http://www.w3.org/2000/svg" aria-hidden="true" fill="currentColor" focusable="false" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 32 32"><path d="M28,10V28H10V10H28m0-2H10a2,2,0,0,0-2,2V28a2,2,0,0,0,2,2H28a2,2,0,0,0,2-2V10a2,2,0,0,0-2-2Z" transform="translate(0)"></path><path d="M4,18H2V4A2,2,0,0,1,4,2H18V4H4Z" transform="translate(0)"></path><rect fill="none" width="32" height="32"></rect></svg> <div class="absolute pointer-events-none transition-opacity bg-black text-white py-1 px-2 leading-tight rounded font-normal shadow left-1/2 top-full transform -translate-x-1/2 translate-y-2 opacity-0"><div class="absolute bottom-full left-1/2 transform -translate-x-1/2 w-0 h-0 border-black border-4 border-t-0" style="border-left-color: transparent; border-right-color: transparent; "></div> Copied</div></button></div> <pre class="language-python "><!-- HTML_TAG_START --><span class="hljs-keyword">import</span> matplotlib.pyplot <span class="hljs-keyword">as</span> plt
<span class="hljs-keyword">import</span> numpy <span class="hljs-keyword">as</span> np
<span class="hljs-comment"># Convert similarities tensor to numpy array</span>
sim_scores = similarities.cpu().numpy() <span class="hljs-keyword">if</span> <span class="hljs-built_in">hasattr</span>(similarities, <span class="hljs-string">&quot;cpu&quot;</span>) <span class="hljs-keyword">else</span> similarities.numpy()
<span class="hljs-comment"># Create paragraph indices</span>
para_indices = np.arange(<span class="hljs-number">1</span>, <span class="hljs-built_in">len</span>(sim_scores) + <span class="hljs-number">1</span>)
<span class="hljs-comment"># Assign colors based on quality thresholds</span>
colors = []
<span class="hljs-keyword">for</span> score <span class="hljs-keyword">in</span> sim_scores:
<span class="hljs-keyword">if</span> score &gt;= <span class="hljs-number">0.85</span>:
colors.append(<span class="hljs-string">&quot;#2ecc71&quot;</span>) <span class="hljs-comment"># Green for high quality</span>
<span class="hljs-keyword">elif</span> score &gt;= <span class="hljs-number">0.75</span>:
colors.append(<span class="hljs-string">&quot;#f39c12&quot;</span>) <span class="hljs-comment"># Orange for medium quality</span>
<span class="hljs-keyword">else</span>:
colors.append(<span class="hljs-string">&quot;#e74c3c&quot;</span>) <span class="hljs-comment"># Red for low quality</span>
<span class="hljs-comment"># Create the figure with two subplots</span>
fig, (ax1, ax2) = plt.subplots(<span class="hljs-number">2</span>, <span class="hljs-number">1</span>, figsize=(<span class="hljs-number">14</span>, <span class="hljs-number">10</span>), gridspec_kw={<span class="hljs-string">&quot;height_ratios&quot;</span>: [<span class="hljs-number">3</span>, <span class="hljs-number">1</span>]})
<span class="hljs-comment"># Main plot: Bar chart of similarities</span>
bars = ax1.bar(para_indices, sim_scores, color=colors, alpha=<span class="hljs-number">0.7</span>, edgecolor=<span class="hljs-string">&quot;black&quot;</span>, linewidth=<span class="hljs-number">0.5</span>)
<span class="hljs-comment"># Add threshold lines</span>
ax1.axhline(y=<span class="hljs-number">0.85</span>, color=<span class="hljs-string">&quot;#2ecc71&quot;</span>, linestyle=<span class="hljs-string">&quot;--&quot;</span>, linewidth=<span class="hljs-number">2</span>, alpha=<span class="hljs-number">0.5</span>, label=<span class="hljs-string">&quot;High Quality (≥0.85)&quot;</span>)
ax1.axhline(y=<span class="hljs-number">0.75</span>, color=<span class="hljs-string">&quot;#f39c12&quot;</span>, linestyle=<span class="hljs-string">&quot;--&quot;</span>, linewidth=<span class="hljs-number">2</span>, alpha=<span class="hljs-number">0.5</span>, label=<span class="hljs-string">&quot;Medium Quality (≥0.75)&quot;</span>)
<span class="hljs-comment"># Add average line</span>
avg_sim = np.mean(sim_scores)
ax1.axhline(y=avg_sim, color=<span class="hljs-string">&quot;blue&quot;</span>, linestyle=<span class="hljs-string">&quot;:&quot;</span>, linewidth=<span class="hljs-number">2</span>, alpha=<span class="hljs-number">0.7</span>, label=<span class="hljs-string">f&quot;Average (<span class="hljs-subst">{avg_sim:<span class="hljs-number">.3</span>f}</span>)&quot;</span>)
<span class="hljs-comment"># Formatting</span>
ax1.set_xlabel(<span class="hljs-string">&quot;Paragraph Number&quot;</span>, fontsize=<span class="hljs-number">12</span>, fontweight=<span class="hljs-string">&quot;bold&quot;</span>)
ax1.set_ylabel(<span class="hljs-string">&quot;Similarity Score&quot;</span>, fontsize=<span class="hljs-number">12</span>, fontweight=<span class="hljs-string">&quot;bold&quot;</span>)
ax1.set_title(
<span class="hljs-string">f&quot;Paragraph-Level Translation Quality\n<span class="hljs-subst">{selected_title_original}</span><span class="hljs-subst">{selected_title_translation}</span>&quot;</span>,
fontsize=<span class="hljs-number">14</span>,
fontweight=<span class="hljs-string">&quot;bold&quot;</span>,
pad=<span class="hljs-number">20</span>,
)
ax1.set_ylim(<span class="hljs-number">0</span>, <span class="hljs-number">1.05</span>)
ax1.grid(<span class="hljs-literal">True</span>, alpha=<span class="hljs-number">0.3</span>, axis=<span class="hljs-string">&quot;y&quot;</span>)
ax1.legend(loc=<span class="hljs-string">&quot;upper right&quot;</span>, fontsize=<span class="hljs-number">10</span>)
<span class="hljs-comment"># Bottom plot: Quality distribution pie chart</span>
high_count = np.<span class="hljs-built_in">sum</span>(sim_scores &gt;= <span class="hljs-number">0.85</span>)
medium_count = np.<span class="hljs-built_in">sum</span>((sim_scores &gt;= <span class="hljs-number">0.75</span>) &amp; (sim_scores &lt; <span class="hljs-number">0.85</span>))
low_count = np.<span class="hljs-built_in">sum</span>(sim_scores &lt; <span class="hljs-number">0.75</span>)
quality_counts = [high_count, medium_count, low_count]
quality_labels = [<span class="hljs-string">f&quot;High\n(<span class="hljs-subst">{high_count}</span>)&quot;</span>, <span class="hljs-string">f&quot;Medium\n(<span class="hljs-subst">{medium_count}</span>)&quot;</span>, <span class="hljs-string">f&quot;Low\n(<span class="hljs-subst">{low_count}</span>)&quot;</span>]
quality_colors = [<span class="hljs-string">&quot;#2ecc71&quot;</span>, <span class="hljs-string">&quot;#f39c12&quot;</span>, <span class="hljs-string">&quot;#e74c3c&quot;</span>]
wedges, texts, autotexts = ax2.pie(
quality_counts,
labels=quality_labels,
colors=quality_colors,
autopct=<span class="hljs-string">&quot;%1.1f%%&quot;</span>,
startangle=<span class="hljs-number">90</span>,
textprops={<span class="hljs-string">&quot;fontsize&quot;</span>: <span class="hljs-number">11</span>},
)
ax2.set_title(<span class="hljs-string">&quot;Quality Distribution&quot;</span>, fontsize=<span class="hljs-number">12</span>, fontweight=<span class="hljs-string">&quot;bold&quot;</span>)
plt.tight_layout()
plt.show()
<span class="hljs-comment"># Print summary statistics</span>
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;\nParagraph Similarity Statistics for: <span class="hljs-subst">{selected_title_original}</span>&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;<span class="hljs-subst">{<span class="hljs-string">&#x27;=&#x27;</span> * <span class="hljs-number">60</span>}</span>&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;Total paragraphs: <span class="hljs-subst">{<span class="hljs-built_in">len</span>(sim_scores)}</span>&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;Average similarity: <span class="hljs-subst">{avg_sim:<span class="hljs-number">.4</span>f}</span>&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;Min similarity: <span class="hljs-subst">{np.<span class="hljs-built_in">min</span>(sim_scores):<span class="hljs-number">.4</span>f}</span> (paragraph <span class="hljs-subst">{np.argmin(sim_scores) + <span class="hljs-number">1</span>}</span>)&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;Max similarity: <span class="hljs-subst">{np.<span class="hljs-built_in">max</span>(sim_scores):<span class="hljs-number">.4</span>f}</span> (paragraph <span class="hljs-subst">{np.argmax(sim_scores) + <span class="hljs-number">1</span>}</span>)&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;Std deviation: <span class="hljs-subst">{np.std(sim_scores):<span class="hljs-number">.4</span>f}</span>&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot;\nQuality breakdown:&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; High (≥0.85): <span class="hljs-subst">{high_count}</span>/<span class="hljs-subst">{<span class="hljs-built_in">len</span>(sim_scores)}</span> (<span class="hljs-subst">{<span class="hljs-number">100</span> * high_count / <span class="hljs-built_in">len</span>(sim_scores):<span class="hljs-number">.1</span>f}</span>%)&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; Medium (0.75-0.85): <span class="hljs-subst">{medium_count}</span>/<span class="hljs-subst">{<span class="hljs-built_in">len</span>(sim_scores)}</span> (<span class="hljs-subst">{<span class="hljs-number">100</span> * medium_count / <span class="hljs-built_in">len</span>(sim_scores):<span class="hljs-number">.1</span>f}</span>%)&quot;</span>)
<span class="hljs-built_in">print</span>(<span class="hljs-string">f&quot; Low (&lt;0.75): <span class="hljs-subst">{low_count}</span>/<span class="hljs-subst">{<span class="hljs-built_in">len</span>(sim_scores)}</span> (<span class="hljs-subst">{<span class="hljs-number">100</span> * low_count / <span class="hljs-built_in">len</span>(sim_scores):<span class="hljs-number">.1</span>f}</span>%)&quot;</span>)<!-- HTML_TAG_END --></pre></div> <h2 class="relative group"><a id="next-steps-and-deployment" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#next-steps-and-deployment"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>Next Steps and Deployment</span></h2> <h3 class="relative group"><a id="customization-ideas" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#customization-ideas"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>Customization Ideas</span></h3> <ol data-svelte-h="svelte-1qpfbvp"><li><p><strong>Different Models</strong>: Replace <code>Qwen/Qwen3-Embedding-4B</code> with other embedding models like <code>Qwen/Qwen3-Embedding-8B</code> for potentially better results</p></li> <li><p><strong>Other Books</strong>: Download different works from Project Gutenberg and compare different language pairs</p> <ul><li>“Le tour du monde en 80 jours - Jules Verne” - <a href="https://www.gutenberg.org/ebooks/800.epub.noimages" rel="nofollow">original</a> - <a href="https://www.gutenberg.org/ebooks/103.epub.noimages" rel="nofollow">translation</a></li></ul></li> <li><p><strong>Batch Processing</strong>: Scale to compare many chapters automatically with configurable thresholds</p></li> <li><p><strong>Export Results</strong>: Save correspondence data to JSON or CSV for further analysis</p></li> <li><p><strong>Task Tuning</strong>: Adjust the instruction prompt in <code>embed_texts()</code> to optimize embeddings for your specific use case</p></li></ol> <h3 class="relative group"><a id="performance-notes" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#performance-notes"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>Performance Notes</span></h3> <ul data-svelte-h="svelte-1nu6azc"><li>Embedding generation time scales with the number of texts and text length</li> <li>Similarity computation is fast (matrix multiplication) once embeddings are available</li> <li>For production use, consider caching embeddings to avoid re-computation</li> <li>Batch size in <code>client.embeddings.create()</code> may need adjustment based on endpoint limits</li></ul> <h3 class="relative group"><a id="troubleshooting" class="header-link block pr-1.5 text-lg no-hover:hidden with-hover:absolute with-hover:p-1.5 with-hover:opacity-0 with-hover:group-hover:opacity-100 with-hover:right-full" href="#troubleshooting"><span><svg class="" xmlns="http://www.w3.org/2000/svg" xmlns:xlink="http://www.w3.org/1999/xlink" aria-hidden="true" role="img" width="1em" height="1em" preserveAspectRatio="xMidYMid meet" viewBox="0 0 256 256"><path d="M167.594 88.393a8.001 8.001 0 0 1 0 11.314l-67.882 67.882a8 8 0 1 1-11.314-11.315l67.882-67.881a8.003 8.003 0 0 1 11.314 0zm-28.287 84.86l-28.284 28.284a40 40 0 0 1-56.567-56.567l28.284-28.284a8 8 0 0 0-11.315-11.315l-28.284 28.284a56 56 0 0 0 79.196 79.197l28.285-28.285a8 8 0 1 0-11.315-11.314zM212.852 43.14a56.002 56.002 0 0 0-79.196 0l-28.284 28.284a8 8 0 1 0 11.314 11.314l28.284-28.284a40 40 0 0 1 56.568 56.567l-28.285 28.285a8 8 0 0 0 11.315 11.314l28.284-28.284a56.065 56.065 0 0 0 0-79.196z" fill="currentColor"></path></svg></span></a> <span>Troubleshooting</span></h3> <p data-svelte-h="svelte-146cnjd">If embeddings fail to generate:</p> <ol data-svelte-h="svelte-thfqvk"><li>Verify endpoint is running and accessible</li> <li>Check your Hugging Face token configuration</li> <li>Review endpoint logs for error messages</li></ol> <p></p>
<script>
{
__sveltekit_1mrdwol = {
assets: "/docs/optimum.neuron/pr_1113/en",
base: "/docs/optimum.neuron/pr_1113/en",
env: {}
};
const element = document.currentScript.parentElement;
const data = [null,null];
Promise.all([
import("/docs/optimum.neuron/pr_1113/en/_app/immutable/entry/start.74026e82.js"),
import("/docs/optimum.neuron/pr_1113/en/_app/immutable/entry/app.79f4069f.js")
]).then(([kit, app]) => {
kit.start(app, element, {
node_ids: [0, 18],
data,
form: null,
error: null
});
});
}
</script>

Xet Storage Details

Size:
94.5 kB
·
Xet hash:
bae897af1ac36cd4daeda6bf7f93e45fc12e75b8e3cff5db721f229d79e87835

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.