File size: 6,531 Bytes
71d239c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
"""Read dependency manifests and explain each library in plain English.

A non-coder has no idea what `chromadb` or `axios` is. We parse
requirements.txt / package.json / pyproject.toml and attach a one-line,
jargon-free description from a curated map (with a generic fallback). No model
call needed for the common case.
"""
from __future__ import annotations

import json
import re

from schema import Dependency

# Curated plain-English blurbs for libraries a vibe-coder is likely to have.
# Keys are matched case-insensitively against the bare package name.
KNOWN: dict[str, str] = {
    # AI / LLM
    "openai": "Talks to OpenAI's AI models (like ChatGPT) to generate text.",
    "anthropic": "Talks to Anthropic's Claude AI models to generate text.",
    "langchain": "A toolkit for chaining AI steps together (load β†’ search β†’ answer).",
    "langchain-community": "Extra connectors for LangChain (databases, loaders, tools).",
    "llama-index": "Helps an AI answer questions over your own documents.",
    "transformers": "Runs open-source AI models on your own machine.",
    "sentence-transformers": "Turns text into numbers (embeddings) so it can be searched by meaning.",
    "chromadb": "A 'memory' database that stores text as numbers for fast meaning-search.",
    "faiss-cpu": "A fast search engine for finding similar pieces of text by meaning.",
    "tiktoken": "Counts how many tokens (word-pieces) your text uses for an AI model.",
    "huggingface-hub": "Downloads AI models and datasets from Hugging Face.",
    "gradio": "Builds the web interface you click on, with very little code.",
    "streamlit": "Builds a simple data web app you click on.",
    # Web / API
    "flask": "A small web server β€” handles requests from the browser.",
    "fastapi": "A modern web server for building fast APIs.",
    "django": "A big all-in-one web framework (pages, database, admin).",
    "uvicorn": "Runs FastAPI apps and serves them to the browser.",
    "requests": "Fetches things from the internet (calls other websites/APIs).",
    "httpx": "Fetches things from the internet, with async support.",
    "aiohttp": "Fetches things from the internet without waiting (async).",
    "express": "A small web server for Node.js β€” handles browser requests.",
    "axios": "Fetches things from the internet from the browser or Node.",
    "next": "A React framework for full websites (pages + server).",
    "react": "Builds interactive user interfaces in the browser.",
    "react-dom": "Connects React to the actual web page.",
    "vue": "Builds interactive user interfaces in the browser.",
    # Data
    "pandas": "Works with tables of data (like a spreadsheet in code).",
    "numpy": "Does fast math on big lists of numbers.",
    "pydantic": "Checks that data has the right shape and types.",
    "sqlalchemy": "Talks to databases using Python instead of raw SQL.",
    "psycopg2": "Connects Python to a PostgreSQL database.",
    "psycopg2-binary": "Connects Python to a PostgreSQL database.",
    "pymongo": "Connects Python to a MongoDB database.",
    "redis": "Talks to Redis, a very fast in-memory store/cache.",
    # Files / media
    "pillow": "Opens and edits images.",
    "pypdf": "Reads text out of PDF files.",
    "pypdf2": "Reads text out of PDF files.",
    "python-docx": "Reads and writes Word documents.",
    "beautifulsoup4": "Pulls data out of web pages (HTML).",
    "opencv-python": "Computer vision β€” works with images and video.",
    # Utility
    "python-dotenv": "Loads secret settings from a .env file.",
    "dotenv": "Loads secret settings from a .env file.",
    "tqdm": "Shows a progress bar while something runs.",
    "click": "Builds command-line tools.",
    "pytest": "Runs your automated tests.",
    "modal": "Runs your code on cloud GPUs without managing servers.",
    "lodash": "A grab-bag of handy helper functions for JavaScript.",
    "dotenv-flow": "Loads secret settings from .env files.",
}

# Best-effort 'this looks old / risky' flags for a non-coder.
RISKY = {
    "psycopg2": "Prefer the maintained build 'psycopg2-binary' or 'psycopg[binary]'.",
    "pypdf2": "PyPDF2 is deprecated β€” 'pypdf' is the maintained successor.",
    "request": "Looks like a typo for 'requests'.",
}

_REQ_LINE = re.compile(r"^\s*([A-Za-z0-9_.\-]+)")


def _plain(name: str) -> str:
    return KNOWN.get(name.lower(), f"A library called '{name}'. Used somewhere in the project.")


def _dep(name: str, manifest: str) -> Dependency:
    low = name.lower()
    return Dependency(name=name, manifest=manifest, plain=_plain(name),
                      risky=low in RISKY)


def parse_requirements(text: str, manifest: str = "requirements.txt") -> list[Dependency]:
    out: list[Dependency] = []
    for line in text.splitlines():
        line = line.strip()
        if not line or line.startswith(("#", "-")):
            continue
        m = _REQ_LINE.match(line)
        if m:
            out.append(_dep(m.group(1), manifest))
    return out


def parse_package_json(text: str, manifest: str = "package.json") -> list[Dependency]:
    out: list[Dependency] = []
    try:
        data = json.loads(text)
    except (json.JSONDecodeError, ValueError):
        return out
    for key in ("dependencies", "devDependencies"):
        for name in (data.get(key) or {}):
            out.append(_dep(name, manifest))
    return out


def parse_pyproject(text: str, manifest: str = "pyproject.toml") -> list[Dependency]:
    """Cheap TOML scan β€” avoids a tomllib dependency for one simple need."""
    out: list[Dependency] = []
    # [project] dependencies = ["foo>=1", "bar"]
    for block in re.findall(r"dependencies\s*=\s*\[(.*?)\]", text, re.DOTALL):
        for item in re.findall(r"['\"]([A-Za-z0-9_.\-]+)", block):
            out.append(_dep(item, manifest))
    # [tool.poetry.dependencies]\n foo = "^1"
    poetry = re.search(r"\[tool\.poetry\.dependencies\](.*?)(?:\n\[|\Z)", text, re.DOTALL)
    if poetry:
        for name in re.findall(r"^\s*([A-Za-z0-9_.\-]+)\s*=", poetry.group(1), re.MULTILINE):
            if name.lower() != "python":
                out.append(_dep(name, manifest))
    return out


def parse_manifest(filename: str, text: str) -> list[Dependency]:
    base = filename.lower()
    if base == "requirements.txt":
        return parse_requirements(text, filename)
    if base == "package.json":
        return parse_package_json(text, filename)
    if base == "pyproject.toml":
        return parse_pyproject(text, filename)
    return []