Sentence Similarity
sentence-transformers
English
zephyr
zephyr-rtos
rag
retrieval
faiss
documentation
embedded
qwen
offline
Instructions to use eoinedge/zephyrproject with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- sentence-transformers
How to use eoinedge/zephyrproject with sentence-transformers:
from sentence_transformers import SentenceTransformer model = SentenceTransformer("eoinedge/zephyrproject") sentences = [ "That is a happy person", "That is a happy dog", "That is a very happy person", "Today is a sunny day" ] embeddings = model.encode(sentences) similarities = model.similarity(embeddings, embeddings) print(similarities.shape) # [4, 4] - Notebooks
- Google Colab
- Kaggle
File size: 7,021 Bytes
41a3668 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 181 182 183 | """Fetch the Zephyr documentation source.
Zephyr publishes no llms.txt — https://docs.zephyrproject.org/llms.txt and
.../latest/llms.txt both 404 — so there is no pre-flattened corpus to pull. The
options were to crawl the rendered HTML from sitemap.xml or to read the RST the
docs are built from. This reads the RST.
Rendered HTML carries navigation, breadcrumbs, version banners and the whole
Doxygen API surface. All of that lands in chunks and competes with prose during
retrieval. The RST is the authoritative text, it is versioned and diffable, and
it is the same tree the upstream llms.txt generator reads — so the corpus here
and the corpus a future llms.txt would produce stay in agreement.
A sparse, blobless, shallow clone keeps this to the doc tree rather than
dragging down a full Zephyr history for a few thousand text files.
python scripts/fetch_docs.py # latest main
python scripts/fetch_docs.py --ref v4.1.0 # a release tag
python scripts/fetch_docs.py --keep-clone # leave the checkout in place
"""
from __future__ import annotations
import argparse
import shutil
import subprocess
import sys
from pathlib import Path
# The default Windows console is cp1252 and raises UnicodeEncodeError on any
# non-Latin-1 character. These scripts print document titles and paths straight
# from the Zephyr tree, which is full of them.
if hasattr(sys.stdout, "reconfigure"):
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
sys.stderr.reconfigure(encoding="utf-8", errors="replace")
REPO = "https://github.com/zephyrproject-rtos/zephyr.git"
ROOT = Path(__file__).resolve().parent.parent
DEFAULT_CLONE = ROOT / "data" / "zephyr-src"
DEFAULT_OUT = ROOT / "data" / "raw_docs"
# Directories under doc/ that hold build machinery or binary assets rather than
# documentation prose.
#
# NOTE: doc/build/ is NOT one of them. It is the "Build and Configuration
# System" section — devicetree, Kconfig, west — and it was excluded here on the
# assumption that the name meant build output. Retrieval testing caught it: a
# query about devicetree bindings returned pinctrl and stepper pages because
# every canonical bindings document had been dropped. Zephyr's doc build writes
# to _build/, not build/.
SKIP_DIRS = {
"_doxygen",
"_extensions",
"_scripts",
"_static",
"_templates",
"_build",
"images",
}
def run(cmd: list[str], cwd: Path | None = None, check: bool = True) -> int:
result = subprocess.run(cmd, cwd=cwd, text=True)
if check and result.returncode != 0:
raise SystemExit(f"command failed ({result.returncode}): {' '.join(cmd)}")
return result.returncode
def sparse_clone(clone_dir: Path, ref: str) -> None:
"""Shallow + blobless + sparse: only doc/, only one commit."""
# An empty leftover directory is not a checkout. Testing `exists()` sent the
# reuse path at a directory with no .git in it, which then failed on fetch
# and again on removal, reporting a file lock that was never the problem.
if (clone_dir / ".git").exists():
# A --branch clone sets a narrow fetch refspec, so a bare `git fetch
# origin main` fails with "couldn't find remote ref". Ask for the ref
# explicitly, and treat the cache as disposable if anything goes wrong:
# it is a checkout of someone else's repository, not state worth
# rescuing.
print(f"Reusing existing checkout at {clone_dir}")
refspec = f"+refs/heads/{ref}:refs/remotes/origin/{ref}"
ok = run(["git", "fetch", "--depth", "1", "origin", refspec], cwd=clone_dir, check=False)
if ok == 0:
ok = run(["git", "checkout", "-f", "FETCH_HEAD"], cwd=clone_dir, check=False)
if ok == 0:
return
print(" cached checkout is unusable - re-cloning")
shutil.rmtree(clone_dir, ignore_errors=True)
if clone_dir.exists():
raise SystemExit(
f"could not remove {clone_dir} (a file may be locked). Delete it and retry."
)
clone_dir.parent.mkdir(parents=True, exist_ok=True)
print(f"Cloning {REPO} ({ref}, doc/ only) -> {clone_dir}")
run(
[
"git",
"clone",
"--depth",
"1",
"--filter=blob:none",
"--sparse",
"--branch",
ref,
REPO,
str(clone_dir),
]
)
run(["git", "sparse-checkout", "set", "doc"], cwd=clone_dir)
def collect(clone_dir: Path, out_dir: Path) -> tuple[int, int]:
"""Copy documentation sources out of the checkout, flattened by path."""
doc_root = clone_dir / "doc"
if not doc_root.is_dir():
raise SystemExit(f"no doc/ directory in {clone_dir}")
if out_dir.exists():
shutil.rmtree(out_dir)
out_dir.mkdir(parents=True)
copied = 0
total_bytes = 0
for path in sorted(doc_root.rglob("*")):
if not path.is_file() or path.suffix.lower() not in {".rst", ".md", ".txt"}:
continue
relative = path.relative_to(doc_root)
if any(part in SKIP_DIRS for part in relative.parts):
continue
# Flatten so the source path survives as the filename. Retrieval cites
# the file, and "kernel/services/threads.rst" is a far more useful
# citation than "threads.rst" repeated across a dozen subsystems.
flat = str(relative).replace("\\", "/").replace("/", "__")
destination = out_dir / flat
destination.write_bytes(path.read_bytes())
copied += 1
total_bytes += destination.stat().st_size
return copied, total_bytes
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--ref", default="main", help="branch or tag to fetch (default: main)")
parser.add_argument("--clone", type=Path, default=DEFAULT_CLONE)
parser.add_argument("--out", type=Path, default=DEFAULT_OUT)
parser.add_argument(
"--keep-clone",
action="store_true",
help="keep the sparse checkout so a later run can update instead of re-cloning",
)
args = parser.parse_args()
sparse_clone(args.clone, args.ref)
copied, total_bytes = collect(args.clone, args.out)
revision = subprocess.run(
["git", "rev-parse", "--short", "HEAD"],
cwd=args.clone,
text=True,
capture_output=True,
).stdout.strip()
# Provenance travels with the corpus. An index built from an unknown commit
# cannot be reproduced or explained later.
(args.out / "_SOURCE.txt").write_text(
f"repository: {REPO}\nref: {args.ref}\ncommit: {revision}\nfiles: {copied}\n",
encoding="utf-8",
)
if not args.keep_clone:
shutil.rmtree(args.clone, ignore_errors=True)
print(f"\n{copied} documents ({total_bytes / 1_048_576:.1f} MB) -> {args.out}")
print(f"Zephyr {args.ref} @ {revision}")
return 0
if __name__ == "__main__":
sys.exit(main())
|