| """Lightweight structure for the non-code-ish files: HTML, CSS, JSON, YAML. |
| |
| These rarely have "functions", but they do reference other files (an HTML page |
| loads scripts and stylesheets) — and those references are real edges in the |
| project graph, so we capture them. |
| """ |
| from __future__ import annotations |
|
|
| import re |
|
|
| from schema import Symbol |
|
|
| _HTML_SRC = re.compile(r"""<(?:script|link|img|a)\b[^>]*?\b(?:src|href)\s*=\s*['"]([^'"]+)['"]""", |
| re.IGNORECASE) |
| _CSS_IMPORT = re.compile(r"""@import\s+(?:url\()?['"]([^'"]+)['"]""") |
| _CSS_RULE = re.compile(r"([.#][A-Za-z_][\w-]*)\s*\{") |
|
|
|
|
| def parse(text: str, lang: str) -> tuple[list[Symbol], list[str]]: |
| if lang == "html": |
| return _parse_html(text) |
| if lang == "css": |
| return _parse_css(text) |
| |
| return [], [] |
|
|
|
|
| def _parse_html(text: str) -> tuple[list[Symbol], list[str]]: |
| refs = [r for r in _HTML_SRC.findall(text) |
| if not r.startswith(("http://", "https://", "//", "data:", "#", "mailto:"))] |
| seen: set[str] = set() |
| uniq = [r for r in refs if not (r in seen or seen.add(r))] |
| return [], uniq |
|
|
|
|
| def _parse_css(text: str) -> tuple[list[Symbol], list[str]]: |
| selectors = _CSS_RULE.findall(text)[:12] |
| symbols = [Symbol(kind="style", name=s) for s in selectors] |
| return symbols, list(dict.fromkeys(_CSS_IMPORT.findall(text))) |
|
|