""" Prepare external drug markdown files into DeepMed RAG-ready dataset format. Usage: python prepare_rag_drug_data.py --source "e:/AI/Thuốc" --target "e:/AI/DeepMed/backend/data" This script: 1) Copies and normalizes files from: - /Thuốc nội bộ -> /thông tin thuốc nội bộ - /cảnh giác dược -> /cảnh giác dược 2) Cleans markdown noise (widget/image-only lines, redundant separators) 3) Ensures each drug markdown begins with: # Hoạt chất: 4) Optionally removes long hash suffixes in filenames. """ from __future__ import annotations import argparse import hashlib import os import re import shutil from dataclasses import dataclass from pathlib import Path from typing import Dict, Iterable, List, Optional, Tuple HASH_SUFFIX_RE = re.compile(r"\s+[0-9a-f]{20,}(?=\.md$)", re.IGNORECASE) MULTI_SPACE_RE = re.compile(r"[ \t]{2,}") EMPTY_LINE_RE = re.compile(r"\n{3,}") HEADING_RE = re.compile(r"^#\s+(.+)$", re.MULTILINE) ACTIVE_RE = re.compile(r"^\s*Hoạt\s*chất\s*:\s*(.+)$", re.IGNORECASE | re.MULTILINE) WIDGET_LINK_RE = re.compile(r"^\s*\[https?://widgetbox\.app/.*\]\(https?://widgetbox\.app/.*\)\s*$", re.IGNORECASE) IMAGE_LINE_RE = re.compile(r"^\s*!\[[^\]]*\]\([^\)]*\)\s*$") HORIZONTAL_RULE_RE = re.compile(r"^\s*---\s*$") @dataclass class SyncStats: copied: int = 0 skipped: int = 0 errors: int = 0 def _clean_filename(name: str, remove_hash_suffix: bool = True) -> str: cleaned = name.strip() if remove_hash_suffix and cleaned.lower().endswith(".md"): cleaned = HASH_SUFFIX_RE.sub("", cleaned) cleaned = MULTI_SPACE_RE.sub(" ", cleaned) return cleaned def _safe_output_path(dest_dir: Path, desired_name: str, content: str) -> Path: out = dest_dir / desired_name if not out.exists(): return out digest = hashlib.md5(content.encode("utf-8", errors="ignore")).hexdigest()[:8] stem = out.stem suffix = out.suffix return dest_dir / f"{stem}__{digest}{suffix}" def _normalize_newlines(text: str) -> str: return text.replace("\r\n", "\n").replace("\r", "\n") def _strip_noise_lines(text: str) -> str: lines = [] for raw in text.split("\n"): line = raw.replace("\xa0", " ").rstrip() if WIDGET_LINK_RE.match(line): continue if IMAGE_LINE_RE.match(line): continue if HORIZONTAL_RULE_RE.match(line): continue lines.append(line) normalized = "\n".join(lines) normalized = EMPTY_LINE_RE.sub("\n\n", normalized) return normalized.strip() def _extract_heading(content: str) -> Optional[str]: m = HEADING_RE.search(content) if not m: return None return m.group(1).strip() def _extract_active_ingredient(content: str) -> Optional[str]: m = ACTIVE_RE.search(content) if not m: return None return m.group(1).strip() def _ensure_rag_header(content: str, fallback_name: str) -> str: heading = _extract_heading(content) or fallback_name active = _extract_active_ingredient(content) or "Chưa cập nhật" body = content body = HEADING_RE.sub("", body, count=1).lstrip("\n") body = ACTIVE_RE.sub("", body, count=1).lstrip("\n") header = f"# {heading}\n\nHoạt chất: {active}\n" out = f"{header}\n{body.strip()}\n" out = EMPTY_LINE_RE.sub("\n\n", out) return out def _read_text(path: Path) -> str: return path.read_text(encoding="utf-8", errors="ignore") def _write_text(path: Path, content: str) -> None: path.parent.mkdir(parents=True, exist_ok=True) path.write_text(content, encoding="utf-8", newline="\n") def _iter_files(root: Path) -> Iterable[Path]: for p in root.rglob("*"): if p.is_file() and p.suffix.lower() in {".md", ".pdf", ".docx", ".txt", ".csv", ".xlsx", ".xls"}: yield p def _normalize_markdown_for_drug(md_text: str, fallback_drug_name: str) -> str: text = _normalize_newlines(md_text) text = _strip_noise_lines(text) text = _ensure_rag_header(text, fallback_drug_name) return text def _sync_folder( source: Path, target: Path, remove_hash_suffix: bool, drug_mode: bool, flatten_tree: bool = False, ) -> SyncStats: stats = SyncStats() if not source.exists(): print(f"[WARN] Source folder not found: {source}") return stats target.mkdir(parents=True, exist_ok=True) seen_names: Dict[str, int] = {} for src in _iter_files(source): try: rel = src.relative_to(source) clean_name = _clean_filename(src.name, remove_hash_suffix=remove_hash_suffix) base_target_dir = target if flatten_tree else (target / rel.parent) base_target_dir.mkdir(parents=True, exist_ok=True) if src.suffix.lower() == ".md": raw = _read_text(src) if drug_mode: fallback = Path(clean_name).stem cleaned = _normalize_markdown_for_drug(raw, fallback) else: cleaned = _strip_noise_lines(_normalize_newlines(raw)) + "\n" out_path = _safe_output_path(base_target_dir, clean_name, cleaned) if out_path.exists(): existing = _read_text(out_path) if existing == cleaned: stats.skipped += 1 continue _write_text(out_path, cleaned) else: out_path = base_target_dir / clean_name if out_path.exists() and out_path.stat().st_size == src.stat().st_size: stats.skipped += 1 continue shutil.copy2(src, out_path) seen_names[out_path.name] = seen_names.get(out_path.name, 0) + 1 stats.copied += 1 except Exception as exc: stats.errors += 1 print(f"[ERROR] Failed: {src} -> {exc}") dup_count = sum(1 for n in seen_names.values() if n > 1) if dup_count: print(f"[WARN] Found {dup_count} duplicate output names; hash suffix applied.") return stats def run(source_root: Path, target_root: Path, remove_hash_suffix: bool = True) -> Tuple[SyncStats, SyncStats]: source_drug = source_root / "Thuốc nội bộ" source_alert = source_root / "cảnh giác dược" target_drug = target_root / "thông tin thuốc nội bộ" target_alert = target_root / "cảnh giác dược" print("=" * 72) print(f"Source root: {source_root}") print(f"Target root: {target_root}") print("=" * 72) drug_stats = _sync_folder( source=source_drug, target=target_drug, remove_hash_suffix=remove_hash_suffix, drug_mode=True, ) alert_stats = _sync_folder( source=source_alert, target=target_alert, remove_hash_suffix=remove_hash_suffix, drug_mode=False, flatten_tree=True, ) print("=" * 72) print("Sync completed") print( "Drug files: copied={0.copied}, skipped={0.skipped}, errors={0.errors} | " "Alert files: copied={1.copied}, skipped={1.skipped}, errors={1.errors}".format( drug_stats, alert_stats ) ) print("=" * 72) return drug_stats, alert_stats def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser(description="Prepare Thuốc files for DeepMed RAG") parser.add_argument( "--source", default=r"e:/AI/Thuốc", help="Path to external Thuốc folder", ) parser.add_argument( "--target", default=r"e:/AI/DeepMed/backend/data", help="Path to DeepMed backend data folder", ) parser.add_argument( "--keep-hash-suffix", action="store_true", help="Keep long hash suffixes in markdown filenames", ) return parser.parse_args() def main() -> None: args = parse_args() source_root = Path(args.source).resolve() target_root = Path(args.target).resolve() run( source_root=source_root, target_root=target_root, remove_hash_suffix=not args.keep_hash_suffix, ) if __name__ == "__main__": main()