NEXORA / nexora /coding.py
devildasdf's picture
Release validated NEXORA research prototype, tiny weights and evidence
12496fc verified
Raw History Blame Contribute Delete
2.42 kB
"""Bounded source indexing. AST symbols/import edges for Python; lexical for others."""
import ast
from collections import Counter
from pathlib import Path
import re
from .tools import Executor
EXTENSIONS = {".py", ".js", ".ts", ".tsx", ".jsx", ".c", ".h", ".cpp", ".hpp", ".cs", ".java", ".go", ".rs", ".sql", ".sh", ".ps1", ".html", ".css", ".kt", ".swift", ".md", ".json"}
EXCLUDED = {".git", "node_modules", ".venv", ".cache", "private", "__pycache__", "artifacts", "reports"}
def index_repository(executor: Executor, max_files=2000, max_bytes=10_000_000):
if "READ" not in executor.policy.permissions:
raise PermissionError("Repository indexing requires READ")
result, total = [], 0
for path in sorted(executor.root.rglob("*")):
if len(result) >= max_files:
break
rel = path.relative_to(executor.root)
if any(p in EXCLUDED for p in rel.parts) or path.suffix not in EXTENSIONS or not path.is_file():
continue
try:
safe = executor.path(str(rel))
size = safe.stat().st_size
if size > executor.policy.max_file_bytes or total + size > max_bytes:
continue
text = safe.read_text(encoding="utf-8")
except (OSError, UnicodeError, PermissionError):
continue
total += size
symbols, imports = [], []
if path.suffix == ".py":
try:
tree = ast.parse(text)
symbols = [{"name": n.name, "line": n.lineno, "kind": type(n).__name__} for n in ast.walk(tree) if isinstance(n, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef))]
imports = [n.module for n in ast.walk(tree) if isinstance(n, ast.ImportFrom)] + [a.name for n in ast.walk(tree) if isinstance(n, ast.Import) for a in n.names]
except SyntaxError:
pass
result.append({"path": rel.as_posix(), "symbols": symbols, "imports": imports,
"terms": dict(Counter(re.findall(r"[A-Za-z_][A-Za-z_0-9]*", text.lower())))})
return result
def retrieve(index, query, limit=8):
terms = set(re.findall(r"[A-Za-z_][A-Za-z_0-9]*", query.lower()))
scored = [(sum(min(r["terms"].get(t, 0), 5) for t in terms), r) for r in index]
return [{k: v for k, v in r.items() if k != "terms"} for score, r in sorted(scored, key=lambda x: x[0], reverse=True) if score > 0][:limit]