Spaces:
Configuration error
Configuration error
File size: 7,050 Bytes
95eb75a | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 | // deepResearchTool.ts — ARL (Adaptive Research Loop)
//
// Tool autocontenuto che incapsula un loop di ricerca multi-round.
// Chiamato come qualsiasi altro tool dal loop principale ReAct:
// executeToolGated("deep_research", { goal: "..." })
//
// Design:
// - Zero LLM aggiuntivo: refineQuery() è deterministico (token-gap detection)
// - Budget fisso: max 3 round × max 4 URL = ≤12 fetch totali (~15s su iPhone free tier)
// - Depth-2: esplora relevant_links della pagina se coverage < minCoverage
// - Hard timeout 22s: non blocca mai il loop ReAct padre
// - Nessuna modifica ad AgentDispatcher, GraphOrchestrator, agent types
import { toolWebSearch, toolReadPage } from "./webSearchCore";
const ARL_BUDGET = {
maxRounds: 3, // max iterazioni di ricerca
maxUrls: 4, // URL da fetchare per round
maxPages: 9, // max pagine totali fetched (iPhone free tier: ~15s totali)
maxDepth: 2, // profondità link-following
minCoverage: 0.70, // token goal coverage per uscita anticipata (doc spec: 0.7)
timeoutMs: 22_000,
} as const;
const _STOP_WORDS = new Set([
"the","and","for","with","that","this","from","into","have","are","was","were",
"che","del","della","per","con","una","uno","gli","dei","nel","nelle","alla",
"sulle","degli","sono","essere","come","quando","dove","anche","però","quindi",
]);
function _tokenize(txt: string): Set<string> {
return new Set(
txt.toLowerCase().split(/\W+/).filter(t => t.length > 3 && !_STOP_WORDS.has(t))
);
}
/** Calcola la frazione di token-goal presenti nel corpus. */
function _goalCoverage(goal: string, corpus: string): number {
const goalToks = _tokenize(goal);
if (goalToks.size === 0) return 1;
const foundToks = _tokenize(corpus);
const covered = [...goalToks].filter(t => foundToks.has(t)).length;
return covered / goalToks.size;
}
/**
* Genera una query di refinement basata sui token-goal mancanti nel corpus.
* Zero LLM — solo analisi lessicale. Ritorna null se il goal è già coperto.
*/
// BUG-FIX: dedup suffix tokens in refineQuery output
function _refineQuery(goal: string, corpus: string): string | null {
const goalToks = [..._tokenize(goal)];
const foundToks = _tokenize(corpus);
const missing = goalToks.filter(t => !foundToks.has(t));
if (missing.length === 0) return null;
// Escludi dal suffix i token già nei missing (evita duplicati es. "finanziamento finanziamento")
const missingSet = new Set(missing.map(t => t.toLowerCase()));
const suffix = goal.trim().split(/\s+/).slice(-2)
.filter(w => !missingSet.has(w.toLowerCase().replace(/\W/g, "")))
.join(" ");
const raw = `${missing.slice(0, 3).join(" ")} ${suffix}`.trim();
// Dedup finale mantenendo l'ordine
const _seen = new Set<string>();
return raw.split(/\s+/).filter(w => {
const k = w.toLowerCase();
return _seen.has(k) ? false : (_seen.add(k), true);
}).join(" ");
}
/**
* Esegui il loop ARL per raccogliere informazioni sul goal.
*
* @param goal — obiettivo di ricerca in linguaggio naturale
* @returns stringa con corpus strutturato (pronto per sintesi del loop ReAct padre)
*/
export async function executeDeepResearch(goal: string): Promise<string> {
const startMs = Date.now();
const visited = new Set<string>();
let corpus = "";
let round = 0;
let query = goal;
let totalFetched = 0; // budget totale pagine
while (round < ARL_BUDGET.maxRounds && Date.now() - startMs < ARL_BUDGET.timeoutMs) {
// ── Round ${round}: search ────────────────────────────────────────────────
const searchResult = await toolWebSearch(query);
const roundUrls = (searchResult.results ?? [])
.map(r => r.url)
.filter(u => !visited.has(u))
.slice(0, Math.min(ARL_BUDGET.maxUrls, ARL_BUDGET.maxPages - totalFetched));
if (roundUrls.length === 0) break;
// ── Fetch parallelo ───────────────────────────────────────────────────────
const pages = await Promise.all(
roundUrls.map(async url => {
visited.add(url);
try {
const r = await toolReadPage(url);
return { url, text: r.output ?? "", links: r.relevant_links ?? [] };
} catch {
return { url, text: "", links: [] };
}
})
);
totalFetched += pages.length;
// ── Aggiungi al corpus ────────────────────────────────────────────────────
for (const p of pages) {
if (p.text && p.text.length > 50) {
corpus += `\n\n--- [${p.url}] ---\n${p.text.slice(0, 1500)}`;
}
}
// ── Coverage check + page budget ─────────────────────────────────────────
const coverage = _goalCoverage(goal, corpus);
if (coverage >= ARL_BUDGET.minCoverage) break; // obiettivo raggiunto
if (totalFetched >= ARL_BUDGET.maxPages) break; // page budget esaurito
// ── Link following (depth ≤ maxDepth) ─────────────────────────────────────
if (round < ARL_BUDGET.maxDepth) {
const linkUrls = pages
.flatMap(p => p.links.map(l => l.url))
.filter(u => u.startsWith("http") && !visited.has(u))
.slice(0, 2);
for (const lu of linkUrls) {
if (Date.now() - startMs > ARL_BUDGET.timeoutMs * 0.80) break;
visited.add(lu);
try {
const lr = await toolReadPage(lu);
if (lr.output && lr.output.length > 50) {
corpus += `\n\n--- [${lu}] (link) ---\n${lr.output.slice(0, 1000)}`;
}
} catch { /* link irraggiungibile — ignora */ }
}
}
// ── Query refinement per prossimo round ───────────────────────────────────
const refined = _refineQuery(goal, corpus);
if (!refined || refined === query) break; // nessun miglioramento possibile
query = refined;
round++;
}
if (!corpus.trim()) {
return `⚠️ deep_research: nessun contenuto raccolto per "${goal}".`;
}
const coverage = _goalCoverage(goal, corpus);
const elapsed = Math.round((Date.now() - startMs) / 1000);
const sourceCount = visited.size;
// OPT: aumentato da 8000→14000 — permette report più completi su argomenti complessi
// senza superare il context window dell'LLM (14KB ≈ 3500 token, gestibili da tutti i modelli)
const CORPUS_LIMIT = 14_000;
return (
`📚 **Deep Research: "${goal}"**\n` +
`_${sourceCount} fonti · ${round + 1} round · copertura ${Math.round(coverage * 100)}% · ${elapsed}s_\n\n` +
corpus.trim().slice(0, CORPUS_LIMIT)
);
}
|