RAGbot / rag.py
fdbw's picture
Create rag.py
9df46ad verified
Raw
History Blame Contribute Delete
1.89 kB
import os
import faiss
from sentence_transformers import SentenceTransformer
from transformers import AutoTokenizer, AutoModelForCausalLM
DATA_PATH = "data/feng_family.txt"
INDEX_DIR = "index"
FAISS_PATH = os.path.join(INDEX_DIR, "family.faiss")
BLOCKS_PATH = os.path.join(INDEX_DIR, "blocks.txt")
embed_model = SentenceTransformer("BAAI/bge-small-zh-v1.5")
def build_index_if_needed():
if os.path.exists(FAISS_PATH) and os.path.exists(BLOCKS_PATH):
return
with open(DATA_PATH, "r", encoding="utf-8") as f:
text = f.read()
blocks = [b.strip() for b in text.split("--- PERSON ---") if b.strip()]
embeddings = embed_model.encode(blocks, normalize_embeddings=True)
index = faiss.IndexFlatIP(embeddings.shape[1])
index.add(embeddings)
os.makedirs(INDEX_DIR, exist_ok=True)
faiss.write_index(index, FAISS_PATH)
with open(BLOCKS_PATH, "w", encoding="utf-8") as f:
for b in blocks:
f.write(b.replace("\n", " ") + "\n===\n")
# 启动时自动建索引(只会跑一次)
build_index_if_needed()
index = faiss.read_index(FAISS_PATH)
with open(BLOCKS_PATH, "r", encoding="utf-8") as f:
blocks = f.read().split("===")
llm_name = "Qwen/Qwen2.5-1.5B"
tokenizer = AutoTokenizer.from_pretrained(llm_name)
model = AutoModelForCausalLM.from_pretrained(llm_name)
SYSTEM_PROMPT = "你是冯氏家谱机器人,只能根据家谱资料回答,没有记载就说未记载。"
def ask(question: str) -> str:
q_emb = embed_model.encode([question], normalize_embeddings=True)
_, I = index.search(q_emb, 1)
context = blocks[I[0][0]]
prompt = f"{SYSTEM_PROMPT}\n\n【家谱资料】\n{context}\n\n【问题】{question}\n回答:"
inputs = tokenizer(prompt, return_tensors="pt")
outputs = model.generate(**inputs, max_new_tokens=200)
return tokenizer.decode(outputs[0], skip_special_tokens=True)