EdgeAIG's picture
download
raw
4.15 kB
/**
* Service for model-specific token counting and prompt truncation. Tokenization
* depends on the target provider, model, and encoding rules, so this module
* leaves the actual tokenization function to the service implementation.
*
* The `Tokenizer` service can count tokens for raw prompt input and shorten a
* prompt to a token limit by keeping the newest messages that fit. This module
* defines the service tag, the service interface, and a `make` constructor that
* builds a full tokenizer service from a token-counting function.
*
* @since 4.0.0
*/
import * as Context from "../../Context.ts";
import * as Effect from "../../Effect.ts";
import type * as AiError from "./AiError.ts";
import * as Prompt from "./Prompt.ts";
declare const Tokenizer_base: Context.ServiceClass<Tokenizer, "effect/ai/Tokenizer", Service>;
/**
* Service tag for model tokenization services.
*
* **When to use**
*
* Use to access or provide model-specific token counting and prompt truncation
* operations.
*
* **Details**
*
* This tag provides access to tokenization functionality throughout your
* application, enabling token counting and prompt truncation capabilities.
*
* **Example** (Accessing the Tokenizer service)
*
* ```ts
* import { Effect } from "effect"
* import { Tokenizer } from "effect/unstable/ai"
*
* const useTokenizer = Effect.gen(function*() {
* const tokenizer = yield* Tokenizer.Tokenizer
* const tokens = yield* tokenizer.tokenize("Hello, world!")
* return tokens.length
* })
* ```
*
* @category services
* @since 4.0.0
*/
export declare class Tokenizer extends Tokenizer_base {
}
/**
* Tokenizer service interface providing text tokenization and truncation
* operations.
*
* **Details**
*
* This interface defines the core operations for converting text to tokens and
* managing content length within token limits for AI model compatibility.
*
* **Example** (Implementing a custom tokenizer)
*
* ```ts
* import { Effect } from "effect"
* import { Prompt } from "effect/unstable/ai"
* import type { Tokenizer } from "effect/unstable/ai"
*
* const customTokenizer: Tokenizer.Service = {
* tokenize: (input) =>
* Effect.succeed(input.toString().split(" ").map((_, i) => i)),
* truncate: (input, maxTokens) =>
* Effect.succeed(Prompt.make(input.toString().slice(0, maxTokens * 5)))
* }
* ```
*
* @category models
* @since 4.0.0
*/
export interface Service {
/**
* Converts text input into an array of token numbers.
*/
readonly tokenize: (
/**
* The text input to tokenize.
*/
input: Prompt.RawInput) => Effect.Effect<Array<number>, AiError.AiError>;
/**
* Truncates text input to fit within the specified token limit.
*/
readonly truncate: (
/**
* The text input to truncate.
*/
input: Prompt.RawInput,
/**
* Maximum number of tokens to retain.
*/
tokens: number) => Effect.Effect<Prompt.Prompt, AiError.AiError>;
}
/**
* Creates a Tokenizer service implementation from tokenization options.
*
* **Details**
*
* This function constructs a complete Tokenizer service by providing a
* tokenization function. The service handles both tokenization and
* truncation operations using the provided tokenizer.
*
* **Example** (Creating a word tokenizer)
*
* ```ts
* import { Effect } from "effect"
* import { Tokenizer } from "effect/unstable/ai"
*
* // Simple word-based tokenizer
* const wordTokenizer = Tokenizer.make({
* tokenize: (prompt) =>
* Effect.succeed(
* prompt.content
* .flatMap((msg) =>
* typeof msg.content === "string"
* ? msg.content.split(" ")
* : msg.content.flatMap((part) =>
* part.type === "text" ? part.text.split(" ") : []
* )
* )
* .map((_, index) => index)
* )
* })
* ```
*
* @category constructors
* @since 4.0.0
*/
export declare const make: (options: {
readonly tokenize: (content: Prompt.Prompt) => Effect.Effect<Array<number>, AiError.AiError>;
}) => Service;
export {};
//# sourceMappingURL=Tokenizer.d.ts.map

Xet Storage Details

Size:
4.15 kB
·
Xet hash:
8bb6f763517d0c4d4f652efd2ed40237f7d997d3d8ee1e3d5d1aad332aff4a0d

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.