everything-webgpu / types /engine /engine.d.ts
nyaaorick's picture
feat: publish everything-webgpu package, engine source and documentation
1944112 verified
Raw
History Blame Contribute Delete
22.6 kB
export class ScheduledEngine {
/**
* @param {object} opts
* @param {ModelStore | import("./model-store.js").StorageAdapter} opts.store
* a ModelStore, or a bare StorageAdapter to wrap in one
* @param {string | URL} [opts.workerUrl]
* @param {() => Promise<object>} [opts.loadWebLLM]
* @param {boolean} [opts.prebuilt] expose WebLLM's 163 HuggingFace-hosted
* models, downloaded on first load. Default true. Set false for an
* offline-only build: `load()` then resolves registered models and nothing
* else, and an unknown id fails before the WebLLM bundle is even fetched.
*/
constructor({ store, workerUrl, loadWebLLM, prebuilt }?: {
store: ModelStore | import("./model-store.js").StorageAdapter;
workerUrl?: string | URL;
loadWebLLM?: () => Promise<object>;
prebuilt?: boolean;
});
/** The ModelStore, so a host can drive the registry without a second handle. */
get store(): ModelStore;
/**
* `chat.completions.create()`, the WebLLM/OpenAI shape. See `chat.js`.
*
* Built once and cached: callers hold on to `engine.chat.completions` the way
* they did with WebLLM, and a fresh object each access would break that.
*/
get chat(): any;
/**
* `environment()` — the read-only report, with `environment.measure()` on it.
*
* Cached like `chat` so a caller can hold on to it. Writes are `configure()`;
* see `environment.js` for why those are separate verbs.
*/
get environment(): any;
get state(): {
status: string;
modelId: any;
progress: any;
error: any;
pool: {
size: number;
busy: number;
queued: number;
};
/** Model ids with a live pool. `modelId` is whichever of them is current. */
resident: any[];
/** Latest decode probe from an engine worker; see multistep.js. */
decode: any;
};
get hasWebGPU(): boolean;
/**
* @param {(state: object) => void} listener called immediately, then on change
* @returns {() => void} unsubscribe
*/
subscribe(listener: (state: object) => void): () => void;
/** Model ids with a live pool right now. */
get resident(): string[];
/**
* Choose which resident model unaddressed requests go to.
*
* Distinct from `load()` on purpose: this is free and instant, because the
* weights are already up. `load()` is what costs.
*/
use(modelId: any): {
status: string;
modelId: any;
progress: any;
error: any;
pool: {
size: number;
busy: number;
queued: number;
};
/** Model ids with a live pool. `modelId` is whichever of them is current. */
resident: any[];
/** Latest decode probe from an engine worker; see multistep.js. */
decode: any;
};
/** Registered models only — cheap, no bundle load. */
listModels(): Promise<any[]>;
/**
* Everything `load()` would accept, normalised: registered models first, then
* WebLLM's prebuilt list.
*
* Costs a WebLLM bundle fetch when `prebuilt` is on, because the list lives
* inside it. `listModels()` is the cheap call if you only care about what this
* app registered.
*
* @returns {Promise<Array<{modelId: string, source: string, model: string,
* contextWindow?: number, vramRequiredMB?: number, sizeBytes?: number}>>}
*/
listAvailableModels(): Promise<Array<{
modelId: string;
source: string;
model: string;
contextWindow?: number;
vramRequiredMB?: number;
sizeBytes?: number;
}>>;
/**
* What this machine will admit to: WebGPU, adapter, `shader-f16`, the five
* limits that matter, storage quota. Cached — hardware does not change
* mid-session, and `requestAdapter()` is not free.
* @returns {Promise<import("./device.js").DeviceProbe>}
*/
probe(): Promise<import("./device.js").DeviceProbe>;
/**
* Whether a model will run here, before anything is downloaded.
* @param {string} modelId
* @returns {Promise<{ok: boolean, blockers: Array<object>, warnings: Array<object>}>}
*/
canRun(modelId: string): Promise<{
ok: boolean;
blockers: Array<object>;
warnings: Array<object>;
}>;
/**
* Which models this device should actually be asked to run, best first.
*
* The prebuilt list spans 239 MB to 31 GB; this is the answer to the first
* question a developer has and the one they have least basis to answer.
*
* @param {{maxVramMB?: number, needsVision?: boolean, needsToolCalling?: boolean,
* prefer?: "quality" | "speed"}} [opts]
*/
recommendModels({ needsToolCalling, ...opts }?: {
maxVramMB?: number;
needsVision?: boolean;
needsToolCalling?: boolean;
prefer?: "quality" | "speed";
}): Promise<{
ok: boolean;
blockers: Array<{
code: string;
message: string;
}>;
warnings: Array<{
code: string;
message: string;
}>;
model: any;
}[]>;
/**
* Is this model's data on disk, so a load would need no network?
*
* Routes by who knows the keys. We wrote an injected model's artifacts and
* hold the manifest, so `verify()` answers exactly — including a `"partial"`
* verdict WebLLM cannot give. Everything else was fetched by WebLLM, which
* derives the keys as its loader did, so `hasModelInCache` is the answer.
*
* @returns {Promise<"cached" | "partial" | "absent">}
*/
cacheState(modelId: any): Promise<"cached" | "partial" | "absent">;
/**
* Download a model into the cache **without building an engine**.
*
* For warming during onboarding: the bytes land while the user is still
* reading, and the later `load()` is a cache read. WebLLM cannot express this
* — `reload()` instantiates the wasm and needs a GPU before it fetches a
* single shard — so this is ours. See `prefetch.js` for the URL-derivation
* risk and the oracle that closes it.
*
* Needs no WebGPU at all, which is the other half of the point: an app can
* warm the cache on a machine it has not yet decided can run the model.
*
* @param {string} modelId
* @param {{signal?: AbortSignal, onProgress?: Function}} [opts]
*/
prefetch(modelId: string, { signal, onProgress }?: {
signal?: AbortSignal;
onProgress?: Function;
}): Promise<{
modelId: string;
files: number;
bytes: number;
alreadyCached: boolean;
}>;
/**
* Free a model's bytes and **keep the registry entry**, so it stays a model
* this engine knows how to get again — the distinction from
* `store.remove()`, which forgets the URL a remote model would need.
*
* Delegates for remote and prebuilt models: `deleteModelAllInfoInCache` is
* WebLLM's, covers tensors + wasm + config, and is maintained upstream.
*/
evict(modelId: any): Promise<{
freedKeys: number;
}>;
/**
* Forget a model entirely: free its bytes **and** drop the registry entry.
*
* `evict()` first, because that is what knows how to reach the bytes for each
* source — and it has to happen before the record is deleted, since for a
* remote model the record holds the only URL those bytes can be derived from.
* Deleting the entry first would strand them in Cache Storage permanently.
*/
remove(modelId: any): Promise<{
freedKeys: number;
}>;
/**
* Projected decode throughput for a model, in tokens per second.
*
* `basis: "measured"` once anything has actually decoded on this machine —
* the engine then knows its own achieved bandwidth and every projection is
* device-specific. Before that, `basis: "extrapolated"` from a reference
* machine, which is a starting point and says so.
*
* Decode is memory-bandwidth-bound, so this is close to the whole story:
* time per token scales with weight bytes and little else.
*
* @param {string} [modelId] defaults to the current model
*/
estimateSpeed(modelId?: string): Promise<{
tokensPerSecond: number;
basis: "measured" | "extrapolated";
modelBytes: number;
bytesPerSecond: number;
reference?: string;
modelId: string;
}>;
/**
* What is actually switched on right now, as opposed to what the device could
* support.
*
* The distinction matters for KV reuse in particular: `probe().kvReuse` is a
* device capability, but the decision is taken inside the engine worker,
* which is the authority. A caller debugging "why is my second turn slow"
* needs the decision, not the capability.
*/
features(): Promise<{
kvReuse: boolean;
shaderF16: boolean;
decodeSteps: any;
multiStepDecoding: boolean;
engines: number;
maxEngines: any;
resident: string[];
computePassBatching: number;
decode: any;
}>;
/**
* Register a model. Two shapes, one call, and the difference is only where
* the bytes come from:
*
* ```js
* // fetched from a base URL you host — an HF repo, a CDN, your own origin
* await engine.registerModel({
* modelId: "my-model",
* model: "/models/my-model/",
* modelLib: "/models/my-model/my-model-webgpu.wasm",
* });
*
* // read off disk. No network connection at any point, ever.
* await engine.registerModel({ modelId: "my-model", files: entries });
* ```
*
* Both end up as one `model_list` entry that WebLLM's own loader resolves the
* same way — the local one only differs in that its base URL is minted on
* `.invalid` and its cache is populated before the loader ever looks.
*
* That origin is the *mechanism* of the offline guarantee, not a marker of
* it: `.invalid` is reserved by RFC 6761 and can never resolve, so there is
* no code path — no bug, no eviction, no future refactor — by which a local
* model reaches the network. It fails with a DNS error instead.
*
* `files` is `{ path, file }[]`; `filesFromDataTransfer` and
* `filesFromInput` build it from a drop event or a directory picker.
*/
registerModel(spec: any): Promise<any>;
/**
* Bring a model up, whatever form you have it in.
*
* One entry point for all three routes, because from a caller's side "load a
* model" is one intention and having to know which of `load`,
* `registerModel` and `ingestModelFolder` to reach for is a decision the
* library can make for them:
*
* ```js
* load("Llama-3.2-1B-Instruct-q4f16_1-MLC") // prebuilt or registered id
* load("https://huggingface.co/mlc-ai/Foo", { modelLib }) // a URL you host
* load({ model, modelLib }) // the same, explicit
* load({ files }) | load(fileList) | load(dataTransfer) // a folder, no network
* ```
*
* `registerModel` and `ingestModelFolder` remain, unchanged, as the low-level
* primitives — this composes them rather than replacing them.
*
* **A URL always needs `modelLib`.** It is not guessed; see `sources.js` for
* the measurement behind that. **`defer: true`** registers the source and
* stops there, returning the record instead of the state — the manager's
* drop-now-load-later flow.
*
* Additive residency: a model already resident stays resident, so switching
* back to it costs nothing. That is only safe while the weights fit, so
* `keepResident: false` (the default) unloads whatever else is up first —
* the old single-model behaviour, and the safe one on a 16 GB machine.
* Pass `keepResident: true` to hold both, having checked the budget yourself
* with `canRun()`.
*
* @param {string | object} src an id, a URL, `{model, modelLib}`, or a folder
* @param {{keepResident?: boolean, signal?: AbortSignal, defer?: boolean,
* id?: string, modelLib?: string, modelType?: string, contextWindow?: number,
* vramRequiredMB?: number, onProgress?: Function}} [opts]
* @returns {Promise<object>} the engine state, or the registry record when `defer`
*/
load(src: string | object, opts?: {
keepResident?: boolean;
signal?: AbortSignal;
defer?: boolean;
id?: string;
modelLib?: string;
modelType?: string;
contextWindow?: number;
vramRequiredMB?: number;
onProgress?: Function;
}): Promise<object>;
/**
* Let a model go, at one of two depths.
*
* ```js
* unload() // the current model's VRAM; cached bytes stay
* unload(id) // that model's VRAM
* unload(id, "cache") // and delete its cached bytes, keeping the registry entry
* ```
*
* At `"vram"` the bytes stay on disk, so loading it again costs no network —
* that is what makes switching back cheap, and the difference between this
* and `remove()`.
*
* **A bare `unload()` frees only the current model**, not every resident one.
* `unloadAll()` is the explicit form for that: freeing everything is the more
* destructive of the two readings and should have to be asked for by name.
*
* @param {string} [modelId] defaults to the current model. Omit both this and
* any resident model to no-op.
* @param {"vram"|"cache"} [level]
*/
unload(modelId?: string, level?: "vram" | "cache"): Promise<{
status: string;
modelId: any;
progress: any;
error: any;
pool: {
size: number;
busy: number;
queued: number;
};
/** Model ids with a live pool. `modelId` is whichever of them is current. */
resident: any[];
/** Latest decode probe from an engine worker; see multistep.js. */
decode: any;
}>;
/** Unload every resident model. */
unloadAll(): Promise<{
status: string;
modelId: any;
progress: any;
error: any;
pool: {
size: number;
busy: number;
queued: number;
};
/** Model ids with a live pool. `modelId` is whichever of them is current. */
resident: any[];
/** Latest decode probe from an engine worker; see multistep.js. */
decode: any;
}>;
/**
* One completion.
*
* Named `complete` rather than `chat` so `engine.chat.completions.create()`
* — the WebLLM-shaped facade, Phase 2 — can take that name without a rename.
*
* @param {CompletionRequest} payload
* @param {(delta: string) => void} [onChunk] called per streamed text delta
* @returns {Promise<CompletionResult>}
*/
complete(payload: CompletionRequest, onChunk?: (delta: string) => void): Promise<CompletionResult>;
/**
* `complete()`, but the callback receives WebLLM's chunk verbatim.
*
* Exists so the `chat.completions.create()` facade can pass chunks straight
* through instead of rebuilding an envelope — which is what dropped
* `tool_calls`, flattened `logprobs` and restamped `created`.
*
* @param {CompletionRequest} payload
* @param {(chunk: object) => void} [onRawChunk]
* @returns {Promise<CompletionResult & {toolCalls?: Array<object>}>}
*/
completeRaw(payload: CompletionRequest, onRawChunk?: (chunk: object) => void): Promise<CompletionResult & {
toolCalls?: Array<object>;
}>;
/**
* One question, one answer, nothing kept.
*
* ```js
* const answer = await engine.ask("Summarise this in one line:\n" + doc);
* ```
*
* @param {string | Array<object>} input
* @param {object} [opts] anything `complete()` takes, plus `onDelta` to stream
* @returns {Promise<string>}
*/
ask(input: string | Array<object>, opts?: object): Promise<string>;
/**
* A multi-turn conversation that keeps its own history.
*
* ```js
* const chat = engine.conversation({ system: "You are terse." });
* await chat.say("hello");
* await chat.say("and again?"); // remembers
* ```
*
* @param {object} [opts] `system`, `keep`, plus `complete()` defaults
*/
conversation(opts?: object): {
readonly messages: any[];
readonly length: number;
say(content: string, onDelta?: (delta: string) => void): Promise<{
text: string;
finishReason: "length" | "stop" | "abort";
}>;
reset(): /*elided*/ any;
restore(messages: any): /*elided*/ any;
};
/**
* Ghost text, with the debounce/supersede/drop-if-stale discipline built in
* and the prompt left to you.
*
* ```js
* const ghost = engine.ghostText({ prompt: (before) => `Continue:\n${before}` });
* editor.on("input", async () => {
* const hint = await ghost.suggest(editor.textBefore());
* if (hint !== null) render(hint); // null means a newer keystroke won
* });
* editor.on("blur", () => ghost.cancel());
* ```
*
* @param {object} opts must include `prompt`
*/
ghostText(opts: object): {
suggest(context: any): Promise<string | null>;
cancel(): number;
};
/**
* Embed text into vectors, through the same scheduler as everything else.
*
* ```js
* const [vector] = await engine.embed("a sentence", { modelId: EMBED_MODEL });
* const vectors = await engine.embed(["one", "two"], { modelId: EMBED_MODEL });
* ```
*
* **Needs an embedding model**, not a chat model — `snowflake-arctic-embed-*`
* in WebLLM's prebuilt list, from 239 MB. They are separate models, so this
* usually names `modelId` explicitly and holds it resident alongside a chat
* model with `load(id, { keepResident: true })`.
*
* Returns bare vectors because that is what a caller does arithmetic on; the
* OpenAI envelope is available as `embedRaw()` for anyone porting code that
* expects `data[].embedding`.
*
* **A running embedding cannot be interrupted.** Cancellation and preemption
* work by making a decode loop break out; one forward pass has no loop, so a
* `cancel()` that lands after the job starts marks it cancelled but does not
* stop it. Queued embeddings supersede and cancel normally. This is tolerable
* because an embedding is milliseconds where a completion is seconds — but it
* is a weaker guarantee than `complete()` gives, so it is stated rather than
* discovered.
*
* @param {string | string[]} input
* @param {{modelId?: string, task?: string, session?: string,
* priority?: string, preemptible?: boolean, id?: string}} [opts]
* @returns {Promise<number[][]>} one vector per input, in order
*/
embed(input: string | string[], opts?: {
modelId?: string;
task?: string;
session?: string;
priority?: string;
preemptible?: boolean;
id?: string;
}): Promise<number[][]>;
/** `embed()`, returning WebLLM's OpenAI-shaped envelope untouched. */
embedRaw(input: any, opts?: {}): Promise<{
data: any;
usage: any;
}>;
/**
* Independent prompts, fanned across the pool. This is the only way to beat
* the ~10 tok/s single-stream ceiling, so anything embarrassingly parallel
* (translating a page, labelling a list) should arrive here rather than as a
* loop of `complete` calls.
*
* @param {CompletionRequest & {requests: Array<Partial<CompletionRequest>>}} payload
* @param {(item: BatchItem) => void} [onItem] called as each item lands
* @returns {Promise<Array<BatchItem>>}
*/
batch(payload: CompletionRequest & {
requests: Array<Partial<CompletionRequest>>;
}, onItem?: (item: BatchItem) => void): Promise<Array<BatchItem>>;
/**
* Cancels by job id or by session key.
* @param {string} idOrSession
* @returns {number} how many jobs it stopped
*/
cancel(idOrSession: string): number;
/**
* Applies a runtime knob to the running pool and persists it as the default.
*
* `decodeSteps` is the multi-step decode width (AI.md, "Multi-step decoding").
* It takes effect on the next burst — no reload — which is what makes sweeping
* it to find this machine's tick boundary cheap.
*/
configure(patch: any): Promise<{
settings: {
decodeSteps: number;
engineCount: number;
};
engines: number;
}>;
#private;
}
/**
* The OpenAI generation fields WebLLM already speaks, plus the scheduling
* fields that are what this engine adds over calling WebLLM directly.
*/
export type CompletionRequest = {
messages: Array<{
role: string;
content: string;
}>;
/**
* load this model first if it is not the live one
*/
modelId?: string;
/**
* job id; also what `cancel(id)` takes
*/
id?: string;
temperature?: number;
max_tokens?: number;
response_format?: object;
extra_body?: object;
/**
* the unit that owns an engine; a whole batch shares one
*/
task?: string;
/**
* a later job with this key supersedes the earlier one
*/
session?: string;
priority?: "interactive" | "normal" | "background";
/**
* may be interrupted by an `interactive` job
*/
preemptible?: boolean;
};
export type CompletionResult = {
text: string;
usage?: object;
/**
* WebLLM's own values
*/
finishReason?: "stop" | "length" | "abort";
/**
* superseded or explicitly cancelled
*/
cancelled?: true;
/**
* an `interactive` job took the slot; `text` is partial
*/
preempted?: true;
};
export type BatchItem = CompletionRequest & {
index: number;
engineIndex: number;
startedAt: number;
finishedAt: number;
error?: string;
};
import { ModelStore } from "./model-store.js";