FreeLLMAPI / server /src /__tests__ /services /routing-exhaustion.test.ts
Nryn215's picture
Upload folder using huggingface_hub
077865a verified
Raw
History Blame Contribute Delete
6.57 kB
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest';
import { routeRequest, setRoutingStrategy } from '../../services/router.js';
import * as ratelimit from '../../services/ratelimit.js';
import { getDb, initDb } from '../../db/index.js';
import * as crypto from '../../lib/crypto.js';
// Mock ratelimit to control quota availability
vi.mock('../../services/ratelimit.js', async () => {
const actual = await vi.importActual('../../services/ratelimit.js');
return {
...actual,
canMakeRequest: vi.fn(),
canUseTokens: vi.fn(),
isOnCooldown: vi.fn(() => false),
};
});
// Mock crypto to avoid IV errors
vi.mock('../../lib/crypto.js', async () => {
const actual = await vi.importActual('../../lib/crypto.js');
return {
...actual,
decrypt: vi.fn(() => 'mocked-api-key'),
};
});
const ORIGINAL_DEV_MODE = process.env.DEV_MODE;
const ORIGINAL_NODE_ENV = process.env.NODE_ENV;
function restoreEnv() {
if (ORIGINAL_DEV_MODE === undefined) {
delete process.env.DEV_MODE;
} else {
process.env.DEV_MODE = ORIGINAL_DEV_MODE;
}
if (ORIGINAL_NODE_ENV === undefined) {
delete process.env.NODE_ENV;
} else {
process.env.NODE_ENV = ORIGINAL_NODE_ENV;
}
}
describe('Routing Key Exhaustion', () => {
beforeEach(() => {
process.env.DEV_MODE = 'true';
process.env.NODE_ENV = 'test';
initDb(':memory:');
// This suite asserts deterministic key/model fallback mechanics, which are
// strategy-independent — pin the legacy priority order so the bandit's
// score-based reordering (now the default) doesn't pick seeded catalog
// models that share the 'google' platform.
setRoutingStrategy('priority');
const db = getDb();
// Setup: 2 models (Pro and Flash)
// Pro is higher priority (priority 1), Flash is lower (priority 2)
db.prepare("INSERT INTO models (platform, model_id, display_name, intelligence_rank, speed_rank, enabled) VALUES ('google', 'gemini-1.5-pro', 'Pro', 1, 1, 1)").run();
db.prepare("INSERT INTO models (platform, model_id, display_name, intelligence_rank, speed_rank, enabled) VALUES ('google', 'gemini-1.5-flash', 'Flash', 2, 2, 1)").run();
const proId = db.prepare("SELECT id FROM models WHERE model_id = 'gemini-1.5-pro'").get().id;
const flashId = db.prepare("SELECT id FROM models WHERE model_id = 'gemini-1.5-flash'").get().id;
db.prepare("INSERT INTO fallback_config (model_db_id, priority, enabled) VALUES (?, 1, 1)").run(proId);
db.prepare("INSERT INTO fallback_config (model_db_id, priority, enabled) VALUES (?, 2, 1)").run(flashId);
// Setup: 2 keys for Google
db.prepare("INSERT INTO api_keys (platform, label, encrypted_key, iv, auth_tag, status, enabled) VALUES ('google', 'Key A', 'enc', 'iv', 'tag', 'healthy', 1)").run();
db.prepare("INSERT INTO api_keys (platform, label, encrypted_key, iv, auth_tag, status, enabled) VALUES ('google', 'Key B', 'enc', 'iv', 'tag', 'healthy', 1)").run();
vi.clearAllMocks();
});
afterEach(() => {
restoreEnv();
});
it('should skip exhausted Key B and use functional Key A for the same high-priority model', () => {
const db = getDb();
const keys = db.prepare("SELECT id, label FROM api_keys").all();
const keyA = keys.find(k => k.label === 'Key A');
const keyB = keys.find(k => k.label === 'Key B');
// Mock behavior:
// Key B is exhausted (returns false for canMakeRequest)
// Key A is functional (returns true)
(ratelimit.canMakeRequest as any).mockImplementation((platform, modelId, keyId) => {
if (keyId === keyB.id) return false;
if (keyId === keyA.id) return true;
return true;
});
(ratelimit.canUseTokens as any).mockReturnValue(true);
// Act: Route request
const result = routeRequest(100);
// Assert: It should have picked the Pro model despite Key B being exhausted
expect(result.modelId).toBe('gemini-1.5-pro');
expect(result.keyId).toBe(keyA.id);
expect(ratelimit.canMakeRequest).toHaveBeenCalled();
});
it('should throw 429 when every key on every model is exhausted', () => {
(ratelimit.canMakeRequest as any).mockReturnValue(false);
expect(() => routeRequest(100)).toThrow(/All models exhausted/);
});
it('should fall back to Flash when Pro is exhausted but Flash has quota', () => {
(ratelimit.canMakeRequest as any).mockImplementation((_platform: string, modelId: string) => {
if (modelId === 'gemini-1.5-pro') return false;
if (modelId === 'gemini-1.5-flash') return true;
return true;
});
(ratelimit.canUseTokens as any).mockReturnValue(true);
const result = routeRequest(100);
expect(result.modelId).toBe('gemini-1.5-flash');
});
// 404 model-removed handling: a dead model is skipped ENTIRELY for the rest
// of the request instead of burning one fallback attempt per key on the same
// dead route. (PR #111, credits @barbotkonv.)
describe('skipModels (model-level 404 skip)', () => {
it('skips every key of a skipped model and routes to the next model', () => {
const db = getDb();
const proId = db.prepare("SELECT id FROM models WHERE model_id = 'gemini-1.5-pro'").get().id;
// Both keys have quota — without skipModels, Pro would be chosen.
(ratelimit.canMakeRequest as any).mockReturnValue(true);
(ratelimit.canUseTokens as any).mockReturnValue(true);
const result = routeRequest(100, undefined, undefined, false, false, new Set([proId]));
expect(result.modelId).toBe('gemini-1.5-flash');
});
it('throws when every model is in skipModels', () => {
const db = getDb();
const ids = db.prepare('SELECT id FROM models WHERE enabled = 1').all().map((r: any) => r.id);
(ratelimit.canMakeRequest as any).mockReturnValue(true);
(ratelimit.canUseTokens as any).mockReturnValue(true);
expect(() => routeRequest(100, undefined, undefined, false, false, new Set(ids))).toThrow();
});
it('overrides a sticky/preferred model that has been skipped', () => {
const db = getDb();
const proId = db.prepare("SELECT id FROM models WHERE model_id = 'gemini-1.5-pro'").get().id;
(ratelimit.canMakeRequest as any).mockReturnValue(true);
(ratelimit.canUseTokens as any).mockReturnValue(true);
// Sticky session prefers Pro, but Pro 404ed earlier in this request.
const result = routeRequest(100, undefined, proId, false, false, new Set([proId]));
expect(result.modelId).toBe('gemini-1.5-flash');
});
});
});