multi-agent-system / scripts /seed_agent_rules_books.py
firepenguindisopanda
updated with new prompts and new workflow to generate a document
94f31ec
Raw
History Blame Contribute Delete
12.6 kB
#!/usr/bin/env python3
"""Seed Pinecone namespaces with agent-rules-books content.
This script loads software engineering book rules from the agent-rules-books
repository into role-specific Pinecone namespaces for RAG retrieval.
"""
import argparse
import asyncio
import os
from pathlib import Path
from typing import Any
from dotenv import load_dotenv
from langchain_core.documents import Document
from langchain_text_splitters import RecursiveCharacterTextSplitter
from app.core.rag import get_rag_service
from app.core.schemas import TeamRole
load_dotenv()
# Book to agent role mapping
# "all" means the book applies to all agent roles
BOOK_TO_ROLES: dict[str, list[str]] = {
"clean-code": ["all"],
"the-pragmatic-programmer": ["all"],
"domain-driven-design": [
"product_owner",
"solution_architect",
"data_architect",
"business_analyst",
],
"domain-driven-design-distilled": [
"product_owner",
"business_analyst",
],
"implementing-domain-driven-design": [
"solution_architect",
"data_architect",
],
"clean-architecture": [
"solution_architect",
"api_designer",
],
"release-it": [
"devops_architect",
"solution_architect",
],
"refactoring": [
"solution_architect",
"api_designer",
],
"designing-data-intensive-applications": [
"data_architect",
"solution_architect",
],
"code-complete": [
"solution_architect",
"api_designer",
],
"working-effectively-with-legacy-code": [
"solution_architect",
],
"patterns-of-enterprise-application-architecture": [
"solution_architect",
"data_architect",
],
"a-philosophy-of-software-design": [
"solution_architect",
"api_designer",
],
}
# All valid agent roles
ALL_ROLES = [
"product_owner",
"business_analyst",
"solution_architect",
"data_architect",
"security_analyst",
"ux_designer",
"api_designer",
"qa_strategist",
"devops_architect",
]
# Version to use for RAG (mini is recommended)
DEFAULT_VERSION = "mini"
def _get_books_directory() -> Path:
"""Get the agent-rules-books directory path."""
# Assume agent-rules-books is at the same level as multi-agent-system
script_dir = Path(__file__).parent
multi_agent_dir = script_dir.parent
specs_before_code_dir = multi_agent_dir.parent
books_dir = specs_before_code_dir / "agent-rules-books"
if not books_dir.exists():
raise FileNotFoundError(
f"agent-rules-books directory not found at {books_dir}. "
"Please ensure the repository is cloned at the correct location."
)
return books_dir
def _build_splitter() -> RecursiveCharacterTextSplitter:
"""Create text splitter for book content."""
return RecursiveCharacterTextSplitter(
chunk_size=800, # Smaller chunks for book rules
chunk_overlap=100,
add_start_index=True,
separators=["\n\n## ", "\n\n", "\n", ". ", " ", ""],
)
def _parse_book_metadata(file_path: Path) -> dict[str, str]:
"""Extract metadata from book file path."""
book_slug = file_path.parent.name
filename = file_path.stem
# Determine version from filename
if filename.endswith(".mini"):
version = "mini"
book_name = filename.replace(".mini", "")
elif filename.endswith(".nano"):
version = "nano"
book_name = filename.replace(".nano", "")
else:
version = "full"
book_name = filename
# Convert slug to readable name
readable_name = book_slug.replace("-", " ").title()
return {
"book_slug": book_slug,
"book_name": readable_name,
"version": version,
}
def _extract_section_from_content(content: str, chunk_start: int) -> str:
"""Determine which section a chunk belongs to."""
# Look backwards from chunk_start to find the last section header
section_markers = [
"## When to use",
"## Primary bias to correct",
"## Decision rules",
"## Trigger rules",
"## Final checklist",
]
content_before = content[:chunk_start]
last_section = "Introduction"
for marker in section_markers:
if marker in content_before:
last_pos = content_before.rfind(marker)
if last_pos > content_before.rfind(last_section):
last_section = marker.replace("## ", "")
return last_section
async def seed_book_for_role(
book_slug: str,
role_name: str,
version: str = DEFAULT_VERSION,
dry_run: bool = False,
) -> dict[str, Any]:
"""Seed a specific book into a role's namespace."""
result: dict[str, Any] = {
"book": book_slug,
"role": role_name,
"version": version,
"chunks_found": 0,
"chunks_inserted": 0,
"errors": [],
}
try:
role = TeamRole(role_name)
except ValueError:
result["errors"].append(f"Invalid role: {role_name}")
return result
books_dir = _get_books_directory()
book_dir = books_dir / book_slug
if not book_dir.exists():
result["errors"].append(f"Book directory not found: {book_dir}")
return result
# Find the version file
version_file = book_dir / f"{book_slug}.{version}.md"
if not version_file.exists():
# Try without version suffix (for full version)
version_file = book_dir / f"{book_slug}.md"
if not version_file.exists():
result["errors"].append(f"Book file not found: {version_file}")
return result
try:
content = version_file.read_text(encoding="utf-8")
except Exception as exc:
result["errors"].append(f"Error reading {version_file.name}: {exc}")
return result
if not content.strip():
result["errors"].append("Book file is empty")
return result
# Parse metadata
metadata = _parse_book_metadata(version_file)
# Split content into chunks
splitter = _build_splitter()
chunks = splitter.split_text(content)
# Create documents
documents: list[Document] = []
for idx, chunk in enumerate(chunks):
# Determine section from chunk position
chunk_start = content.find(chunk)
section = _extract_section_from_content(content, chunk_start)
doc_metadata = {
"source": version_file.name,
"book_name": metadata["book_name"],
"book_slug": metadata["book_slug"],
"version": metadata["version"],
"section": section,
"rule_index": idx,
"role": role_name,
"file_type": ".md",
"file_path": str(version_file.relative_to(books_dir)),
"chunk_index": idx,
"total_chunks": len(chunks),
"content_type": "engineering_book",
}
documents.append(
Document(
page_content=chunk,
metadata=doc_metadata,
)
)
result["chunks_found"] = len(documents)
if dry_run:
print(f" [DRY RUN] Would insert {len(documents)} chunks")
return result
# Insert into Pinecone
rag_service = get_rag_service()
if not rag_service.is_pinecone_available():
result["errors"].append("Pinecone backend not available")
return result
try:
ids = await rag_service.add_documents(documents=documents, role=role)
result["chunks_inserted"] = len(ids)
except Exception as exc:
result["errors"].append(f"Error inserting documents: {exc}")
return result
async def seed_all_books(
books: list[str] | None = None,
roles: list[str] | None = None,
version: str = DEFAULT_VERSION,
dry_run: bool = False,
) -> int:
"""Seed all books into appropriate role namespaces."""
books_to_seed = books or list(BOOK_TO_ROLES.keys())
roles_to_seed = roles or ALL_ROLES
print(f"Seeding agent-rules-books{' [DRY RUN]' if dry_run else ''}...")
print(f"Version: {version}")
print(f"Books: {len(books_to_seed)}")
print(f"Roles: {len(roles_to_seed)}")
print()
total_chunks = 0
total_inserted = 0
total_errors = 0
operations = 0
for book_slug in books_to_seed:
if book_slug not in BOOK_TO_ROLES:
print(f"Warning: Unknown book '{book_slug}', skipping")
continue
book_roles = BOOK_TO_ROLES[book_slug]
# Expand "all" to all roles
if "all" in book_roles:
target_roles = roles_to_seed
else:
target_roles = [r for r in book_roles if r in roles_to_seed]
if not target_roles:
continue
print(f"Processing: {book_slug}")
print(f" Target roles: {', '.join(target_roles)}")
for role_name in target_roles:
operations += 1
result = await seed_book_for_role(book_slug, role_name, version, dry_run)
total_chunks += result["chunks_found"]
total_inserted += result["chunks_inserted"]
if result["errors"]:
total_errors += len(result["errors"])
for err in result["errors"]:
print(f" [{role_name}] ERROR: {err}")
elif dry_run:
print(f" [{role_name}] Found {result['chunks_found']} chunks")
else:
print(f" [{role_name}] Inserted {result['chunks_inserted']} chunks")
print()
print("=" * 60)
print("SEEDING COMPLETE")
print("=" * 60)
print(f"Total operations: {operations}")
print(f"Total chunks found: {total_chunks}")
if not dry_run:
print(f"Total chunks inserted: {total_inserted}")
if total_errors:
print(f"Total errors: {total_errors}")
return 0 if total_errors == 0 else 1
def _validate_env(dry_run: bool) -> bool:
"""Validate required environment variables."""
if dry_run:
return True
required = ["PINECONE_API_KEY", "PINECONE_INDEX", "NVIDIA_API_KEY"]
missing = [key for key in required if not os.getenv(key)]
if missing:
print(f"Error: missing required environment variables: {', '.join(missing)}")
return False
return True
def main(argv: list[str] | None = None) -> int:
"""Main entry point."""
parser = argparse.ArgumentParser(
description="Seed Pinecone with agent-rules-books content"
)
parser.add_argument(
"--book",
type=str,
help="Specific book slug to seed (e.g., clean-code)",
)
parser.add_argument(
"--role",
type=str,
help="Specific role to seed (e.g., solution_architect)",
)
parser.add_argument(
"--version",
type=str,
default=DEFAULT_VERSION,
choices=["mini", "nano", "full"],
help=f"Book version to use (default: {DEFAULT_VERSION})",
)
parser.add_argument(
"--dry-run",
action="store_true",
help="Preview without inserting into Pinecone",
)
parser.add_argument(
"--list-books",
action="store_true",
help="List available books",
)
parser.add_argument(
"--list-roles",
action="store_true",
help="List available roles",
)
args = parser.parse_args(argv)
if args.list_books:
print("Available books:")
for book in sorted(BOOK_TO_ROLES.keys()):
roles = BOOK_TO_ROLES[book]
print(f" {book}: {', '.join(roles)}")
return 0
if args.list_roles:
print("Available roles:")
for role in sorted(ALL_ROLES):
print(f" {role}")
return 0
if not _validate_env(args.dry_run):
return 1
books = [args.book] if args.book else None
roles = [args.role] if args.role else None
return asyncio.run(
seed_all_books(
books=books,
roles=roles,
version=args.version,
dry_run=args.dry_run,
)
)
if __name__ == "__main__":
raise SystemExit(main())
# Made with Bob