File size: 7,021 Bytes
41a3668
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
"""Fetch the Zephyr documentation source.

Zephyr publishes no llms.txt — https://docs.zephyrproject.org/llms.txt and
.../latest/llms.txt both 404 — so there is no pre-flattened corpus to pull. The
options were to crawl the rendered HTML from sitemap.xml or to read the RST the
docs are built from. This reads the RST.

Rendered HTML carries navigation, breadcrumbs, version banners and the whole
Doxygen API surface. All of that lands in chunks and competes with prose during
retrieval. The RST is the authoritative text, it is versioned and diffable, and
it is the same tree the upstream llms.txt generator reads — so the corpus here
and the corpus a future llms.txt would produce stay in agreement.

A sparse, blobless, shallow clone keeps this to the doc tree rather than
dragging down a full Zephyr history for a few thousand text files.

    python scripts/fetch_docs.py                    # latest main
    python scripts/fetch_docs.py --ref v4.1.0       # a release tag
    python scripts/fetch_docs.py --keep-clone       # leave the checkout in place
"""

from __future__ import annotations

import argparse
import shutil
import subprocess
import sys
from pathlib import Path

# The default Windows console is cp1252 and raises UnicodeEncodeError on any
# non-Latin-1 character. These scripts print document titles and paths straight
# from the Zephyr tree, which is full of them.
if hasattr(sys.stdout, "reconfigure"):
    sys.stdout.reconfigure(encoding="utf-8", errors="replace")
    sys.stderr.reconfigure(encoding="utf-8", errors="replace")

REPO = "https://github.com/zephyrproject-rtos/zephyr.git"
ROOT = Path(__file__).resolve().parent.parent
DEFAULT_CLONE = ROOT / "data" / "zephyr-src"
DEFAULT_OUT = ROOT / "data" / "raw_docs"

# Directories under doc/ that hold build machinery or binary assets rather than
# documentation prose.
#
# NOTE: doc/build/ is NOT one of them. It is the "Build and Configuration
# System" section — devicetree, Kconfig, west — and it was excluded here on the
# assumption that the name meant build output. Retrieval testing caught it: a
# query about devicetree bindings returned pinctrl and stepper pages because
# every canonical bindings document had been dropped. Zephyr's doc build writes
# to _build/, not build/.
SKIP_DIRS = {
    "_doxygen",
    "_extensions",
    "_scripts",
    "_static",
    "_templates",
    "_build",
    "images",
}


def run(cmd: list[str], cwd: Path | None = None, check: bool = True) -> int:
    result = subprocess.run(cmd, cwd=cwd, text=True)
    if check and result.returncode != 0:
        raise SystemExit(f"command failed ({result.returncode}): {' '.join(cmd)}")
    return result.returncode


def sparse_clone(clone_dir: Path, ref: str) -> None:
    """Shallow + blobless + sparse: only doc/, only one commit."""
    # An empty leftover directory is not a checkout. Testing `exists()` sent the
    # reuse path at a directory with no .git in it, which then failed on fetch
    # and again on removal, reporting a file lock that was never the problem.
    if (clone_dir / ".git").exists():
        # A --branch clone sets a narrow fetch refspec, so a bare `git fetch
        # origin main` fails with "couldn't find remote ref". Ask for the ref
        # explicitly, and treat the cache as disposable if anything goes wrong:
        # it is a checkout of someone else's repository, not state worth
        # rescuing.
        print(f"Reusing existing checkout at {clone_dir}")
        refspec = f"+refs/heads/{ref}:refs/remotes/origin/{ref}"
        ok = run(["git", "fetch", "--depth", "1", "origin", refspec], cwd=clone_dir, check=False)
        if ok == 0:
            ok = run(["git", "checkout", "-f", "FETCH_HEAD"], cwd=clone_dir, check=False)
        if ok == 0:
            return
        print("  cached checkout is unusable - re-cloning")
        shutil.rmtree(clone_dir, ignore_errors=True)
        if clone_dir.exists():
            raise SystemExit(
                f"could not remove {clone_dir} (a file may be locked). Delete it and retry."
            )

    clone_dir.parent.mkdir(parents=True, exist_ok=True)
    print(f"Cloning {REPO} ({ref}, doc/ only) -> {clone_dir}")
    run(
        [
            "git",
            "clone",
            "--depth",
            "1",
            "--filter=blob:none",
            "--sparse",
            "--branch",
            ref,
            REPO,
            str(clone_dir),
        ]
    )
    run(["git", "sparse-checkout", "set", "doc"], cwd=clone_dir)


def collect(clone_dir: Path, out_dir: Path) -> tuple[int, int]:
    """Copy documentation sources out of the checkout, flattened by path."""
    doc_root = clone_dir / "doc"
    if not doc_root.is_dir():
        raise SystemExit(f"no doc/ directory in {clone_dir}")

    if out_dir.exists():
        shutil.rmtree(out_dir)
    out_dir.mkdir(parents=True)

    copied = 0
    total_bytes = 0
    for path in sorted(doc_root.rglob("*")):
        if not path.is_file() or path.suffix.lower() not in {".rst", ".md", ".txt"}:
            continue
        relative = path.relative_to(doc_root)
        if any(part in SKIP_DIRS for part in relative.parts):
            continue

        # Flatten so the source path survives as the filename. Retrieval cites
        # the file, and "kernel/services/threads.rst" is a far more useful
        # citation than "threads.rst" repeated across a dozen subsystems.
        flat = str(relative).replace("\\", "/").replace("/", "__")
        destination = out_dir / flat
        destination.write_bytes(path.read_bytes())
        copied += 1
        total_bytes += destination.stat().st_size

    return copied, total_bytes


def main() -> int:
    parser = argparse.ArgumentParser(description=__doc__)
    parser.add_argument("--ref", default="main", help="branch or tag to fetch (default: main)")
    parser.add_argument("--clone", type=Path, default=DEFAULT_CLONE)
    parser.add_argument("--out", type=Path, default=DEFAULT_OUT)
    parser.add_argument(
        "--keep-clone",
        action="store_true",
        help="keep the sparse checkout so a later run can update instead of re-cloning",
    )
    args = parser.parse_args()

    sparse_clone(args.clone, args.ref)
    copied, total_bytes = collect(args.clone, args.out)

    revision = subprocess.run(
        ["git", "rev-parse", "--short", "HEAD"],
        cwd=args.clone,
        text=True,
        capture_output=True,
    ).stdout.strip()

    # Provenance travels with the corpus. An index built from an unknown commit
    # cannot be reproduced or explained later.
    (args.out / "_SOURCE.txt").write_text(
        f"repository: {REPO}\nref: {args.ref}\ncommit: {revision}\nfiles: {copied}\n",
        encoding="utf-8",
    )

    if not args.keep_clone:
        shutil.rmtree(args.clone, ignore_errors=True)

    print(f"\n{copied} documents ({total_bytes / 1_048_576:.1f} MB) -> {args.out}")
    print(f"Zephyr {args.ref} @ {revision}")
    return 0


if __name__ == "__main__":
    sys.exit(main())