Spaces:
Running
Running
| // The engine is the flux-klein.js package (npm). | |
| // It picks one of two pipelines behind one entry point: | |
| // desktop — int8 transformer resident on the GPU (~4 GB), fastest | |
| // mobile — int4 shards streamed through a small GPU ring (~0.6 GB), built to survive the | |
| // iPhone's 2 GB tab cap; also fine on any low-memory device | |
| // This module is the page's view of it: which engine to pick, the cache probe that runs before | |
| // anything else is loaded, and one status line per progress event. | |
| import { isIOS, isMobile } from 'flux-klein.js/platform' | |
| import { formatProgress, progressFraction } from 'flux-klein.js/progress' | |
| import { defaultCache } from 'flux-klein.js/cache' | |
| import { MAX_AREA, MAX_AREA_ONE_THING, MAX_SIDE, maxAreaFor } from 'flux-klein.js/limits' | |
| // onnxruntime-web's wasm runtime as Vite assets, so it is served from here rather than from the | |
| // package's default (the jsDelivr CDN). Both files are subpath exports of onnxruntime-web. | |
| import ortMjs from 'onnxruntime-web/ort-wasm-simd-threaded.mjs?url' | |
| import ortWasm from 'onnxruntime-web/ort-wasm-simd-threaded.wasm?url' | |
| import { HF_BASE } from './hf.js' | |
| export const IOS = isIOS() | |
| export const IS_MOBILE = isMobile() | |
| // ?engine=mobile|desktop overrides the detection (e.g. to try the streaming engine on a laptop). | |
| export function pickEngine() { | |
| const q = new URLSearchParams(location.search).get('engine') | |
| return q === 'mobile' || q === 'desktop' ? q : IS_MOBILE ? 'mobile' : 'desktop' | |
| } | |
| const ROOT = new URL(import.meta.env.BASE_URL, location.href).href // where public/ is served | |
| // The limits the engine will enforce, known before it loads: each side up to MAX_SIDE, and an | |
| // area budget — the phone build on an iPhone (one heavy thing in memory at a time) caps it at | |
| // MAX_AREA_ONE_THING, everything else at what the GPU can bind in one attention buffer | |
| // (flux-klein.js/limits maxAreaFor), MAX_AREA at most. Knowing them up front keeps the size menu | |
| // from offering presets the first run would refuse; the loaded engine confirms them (loadEngine). | |
| const F16 = () => new URLSearchParams(location.search).get('f16') !== '0' | |
| export const limitsFor = (kind, gpuLimits = null) => ({ | |
| maxSide: MAX_SIDE, | |
| maxArea: maxAreaFor(gpuLimits, { | |
| es: F16() ? 2 : 4, | |
| cap: kind === 'mobile' && IOS ? MAX_AREA_ONE_THING : MAX_AREA, | |
| }), | |
| }) | |
| // The same, refined with the adapter's limits. An adapter is a handle, not a device: no GPU | |
| // memory is taken. Falls back to the static guess where WebGPU is missing. | |
| export async function probeLimits(kind) { | |
| try { | |
| const adapter = await navigator.gpu?.requestAdapter() | |
| return limitsFor(kind, adapter?.limits ?? null) | |
| } catch { | |
| return limitsFor(kind) | |
| } | |
| } | |
| // The weights already in this browser, and wiping them — both without loading the engine | |
| // (flux-klein.js/cache reads OPFS directory metadata only: no weights, no GPU device). | |
| export const cacheInfo = () => defaultCache.cacheInfo() | |
| export const clearCache = () => defaultCache.clearCache() | |
| // `onStatus(text, fraction)`: one status line at a time, with a 0…1 fraction while the line | |
| // carries one (download, upload, denoise) or null. The package reports { stage, detail } events | |
| // (flux-klein.js/progress); the wording and the fraction come from there, not from regexes. | |
| export async function loadEngine(kind, onStatus) { | |
| // The package and the tokenizer class come in only now: nothing loads before the first Generate. | |
| const [{ createFluxKlein }, { PreTrainedTokenizer }] = await Promise.all([ | |
| import('flux-klein.js'), | |
| import('@huggingface/transformers'), | |
| ]) | |
| const klein = await createFluxKlein({ | |
| mode: kind, | |
| base: HF_BASE, | |
| tokenizerUrl: ROOT + 'tokenizer/', | |
| f16: F16(), | |
| // Absolute: the mobile build hands these to its ONNX workers, whose base URL is not the page. | |
| wasmPaths: { | |
| mjs: new URL(ortMjs, location.href).href, | |
| wasm: new URL(ortWasm, location.href).href, | |
| }, | |
| PreTrainedTokenizer, | |
| // Threads need cross-origin isolation; inside the hub's iframe that is never granted, and the | |
| // package falls back to one thread on its own. | |
| onEvent: (ev) => onStatus(formatProgress(ev), progressFraction(ev)), | |
| }) | |
| addEventListener('pagehide', () => klein.destroy(), { once: true }) | |
| return { | |
| kind: klein.mode, | |
| klein, | |
| f16: klein.f16, | |
| maxSide: klein.limits.maxSide, // per side 128…maxSide, in steps of 16 | |
| maxArea: klein.limits.maxArea, // the phone build caps the area below maxSide² | |
| cacheInfo: () => klein.cacheInfo(), | |
| clearCache: () => klein.clearCache(), | |
| } | |
| } | |
| // Generate one image. `ref` (ImageBitmap) turns generation into an edit of that image: the package | |
| // cover-fits it to a square, encodes it with the VAE (at 256² unless the output is a 128² or 256² | |
| // square — enc512 overflows ORT-web's shape math) and the prompt describes the change. | |
| export function generate(engine, { prompt, width, height, steps, seed, ref }) { | |
| return engine.klein.generate({ prompt, width, height, steps, seed, reference: ref ?? null }) | |
| } | |