Marek4321's picture
initial commit
999aff1
Raw
History Blame Contribute Delete
4.86 kB
import json
import streamlit as st
from corpus_loader import build_corpus
from dci_engine import search_corpus
from llm_planner import plan_search, synthesize_answer
st.set_page_config(page_title="DCI Agent", page_icon="πŸ”", layout="wide")
st.title("πŸ” DCI Agent")
st.caption("Przeszukuj dokumenty bez embeddingΓ³w i baz wektorowych.")
# ── Sidebar ──────────────────────────────────────────────────────────────────
with st.sidebar:
st.header("Konfiguracja")
api_key = st.text_input("OpenAI API Key", type="password", placeholder="sk-...")
col_btn, col_status = st.columns([1, 2])
with col_btn:
fetch_clicked = st.button("Pobierz modele", disabled=not api_key)
if fetch_clicked:
try:
from openai import OpenAI as _OAI
_models = _OAI(api_key=api_key).models.list()
gpt_ids = sorted(
[m.id for m in _models if "gpt" in m.id and "realtime" not in m.id and "audio" not in m.id],
reverse=True,
)
st.session_state["available_models"] = gpt_ids
with col_status:
st.success(f"{len(gpt_ids)} modeli")
except Exception as e:
with col_status:
st.error(str(e))
_default_models = ["gpt-4o-mini", "gpt-4o"]
_model_list = st.session_state.get("available_models", _default_models)
model = st.selectbox("Model", _model_list)
st.header("Korpus dokumentΓ³w")
uploaded = st.file_uploader(
"Wgraj pliki (.txt / .md / .pdf)",
type=["txt", "md", "pdf"],
accept_multiple_files=True,
)
if st.button("ZaΕ‚aduj korpus", disabled=not uploaded):
import tempfile, os
paths = []
for f in uploaded:
suffix = os.path.splitext(f.name)[1]
with tempfile.NamedTemporaryFile(delete=False, suffix=suffix) as tmp:
tmp.write(f.read())
paths.append((tmp.name, f.name))
corpus = {}
for tmp_path, orig_name in paths:
from corpus_loader import load_file
try:
text = load_file(tmp_path)
corpus[orig_name] = text.splitlines()
except Exception as e:
st.warning(f"Nie można wczytać {orig_name}: {e}")
finally:
os.unlink(tmp_path)
st.session_state["corpus"] = corpus
total = sum(len(v) for v in corpus.values())
st.success(f"ZaΕ‚adowano {len(corpus)} plik(Γ³w), {total} linii.")
# ── Main area ─────────────────────────────────────────────────────────────────
corpus = st.session_state.get("corpus")
if corpus:
st.info(f"Korpus: {len(corpus)} plik(Γ³w) β€” " + ", ".join(corpus.keys()))
question = st.text_area("Pytanie", placeholder="Co chcesz wiedzieć z dokumentów?", height=80)
if st.button("πŸ”Ž Szukaj", type="primary"):
if not api_key:
st.error("Podaj klucz OpenAI API w panelu bocznym.")
elif not corpus:
st.error("Najpierw zaΕ‚aduj korpus w panelu bocznym.")
elif not question.strip():
st.error("Wpisz pytanie.")
else:
from openai import OpenAI
client = OpenAI(api_key=api_key)
with st.status("Planowanie zapytaΕ„...", expanded=True) as status:
plan = plan_search(question, list(corpus.keys()), client)
st.write("Plan:", plan.get("strategy", ""))
status.update(label="Przeszukiwanie korpusu...")
hits = search_corpus(corpus, plan.get("chain_queries", []))
st.write(f"Znaleziono {len(hits)} trafieΕ„.")
status.update(label="Synteza odpowiedzi...")
if hits:
answer = synthesize_answer(question, hits, client)
else:
answer = "Nie znaleziono pasujΔ…cych fragmentΓ³w w korpusie."
status.update(label="Gotowe", state="complete")
st.subheader("OdpowiedΕΊ")
st.markdown(answer)
tab1, tab2 = st.tabs(["Plan wyszukiwania", "Surowe trafienia"])
with tab1:
st.json(plan)
with tab2:
if hits:
rows = [
{
"plik": h["file"],
"linia": h["line_number"],
"treΕ›Δ‡": h["matched_line"].strip(),
"kontekst": " | ".join(l.strip() for l in h["context"]),
}
for h in hits
]
st.dataframe(rows, use_container_width=True)
else:
st.write("Brak trafieΕ„.")