Instructions to use U4RASD/DRU-RE-Yehia-2 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use U4RASD/DRU-RE-Yehia-2 with PEFT:
from peft import PeftModel from transformers import AutoModelForCausalLM base_model = AutoModelForCausalLM.from_pretrained("Navid-AI/Yehia-7B-preview") model = PeftModel.from_pretrained(base_model, "U4RASD/DRU-RE-Yehia-2") - Transformers
How to use U4RASD/DRU-RE-Yehia-2 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="U4RASD/DRU-RE-Yehia-2") messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("U4RASD/DRU-RE-Yehia-2", device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use U4RASD/DRU-RE-Yehia-2 with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "U4RASD/DRU-RE-Yehia-2" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "U4RASD/DRU-RE-Yehia-2", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/U4RASD/DRU-RE-Yehia-2
- SGLang
How to use U4RASD/DRU-RE-Yehia-2 with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "U4RASD/DRU-RE-Yehia-2" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "U4RASD/DRU-RE-Yehia-2", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "U4RASD/DRU-RE-Yehia-2" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "U4RASD/DRU-RE-Yehia-2", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use U4RASD/DRU-RE-Yehia-2 with Docker Model Runner:
docker model run hf.co/U4RASD/DRU-RE-Yehia-2
| #!/usr/bin/env python3 | |
| """Generate Yehia-2 blind-test Codabench artifacts with candidate p(no_relation) >= 0.5. | |
| This script intentionally writes only under: | |
| /root/workspace/DRU-RE-Yehia-2/test_codabench_threshold_0.5 | |
| It uses the copied and patched full-pipeline runner in ../runner_threshold_0_5. | |
| """ | |
| from __future__ import annotations | |
| import argparse | |
| import importlib.util | |
| import json | |
| import os | |
| import shutil | |
| import sys | |
| from pathlib import Path | |
| from typing import Any, Dict, Tuple | |
| from dotenv import load_dotenv | |
| REPO = Path("/root/workspace/DRU-RE-Yehia-2") | |
| WORKSPACE = Path("/root/workspace") | |
| OUTPUT_ROOT = REPO / "test_codabench_threshold_0.5" | |
| REFERENCE_RUNNER = OUTPUT_ROOT / "runner_threshold_0_5" / "run_test.py" | |
| BASE_MODEL_ID = "Navid-AI/Yehia-7B-preview" | |
| BASE_MODEL_REVISION = "b9dda4715eafee7e8090d2c83cfe078d75f4ebb8" | |
| PROMPT_VERSION = "yehia_re_ar_v3_one_token_choice" | |
| METHODS = ("majority_vote", "soft_pool", "max_positive", "top_confidence") | |
| def force_workspace_caches() -> None: | |
| cache_root = WORKSPACE / ".cache" | |
| temp_root = WORKSPACE / "tmp" | |
| for path in (cache_root, temp_root): | |
| path.mkdir(parents=True, exist_ok=True) | |
| os.environ["HF_HOME"] = str(cache_root / "huggingface") | |
| os.environ["HF_HUB_CACHE"] = str(cache_root / "huggingface" / "hub") | |
| os.environ["TRANSFORMERS_CACHE"] = str(cache_root / "huggingface" / "transformers") | |
| os.environ["XDG_CACHE_HOME"] = str(cache_root) | |
| os.environ["TORCH_HOME"] = str(cache_root / "torch") | |
| os.environ["TRITON_CACHE_DIR"] = str(cache_root / "triton") | |
| os.environ["VLLM_CACHE_ROOT"] = str(cache_root / "vllm") | |
| os.environ["VLLM_CONFIG_ROOT"] = str(cache_root / "vllm_config") | |
| os.environ["RAY_TMPDIR"] = str(temp_root / "ray") | |
| os.environ["TMPDIR"] = str(temp_root) | |
| def copy_required_file(source: Path, target: Path) -> None: | |
| if not source.is_file(): | |
| raise FileNotFoundError(source) | |
| target.parent.mkdir(parents=True, exist_ok=True) | |
| shutil.copy2(source, target) | |
| def copytree_replace(source: Path, target: Path) -> None: | |
| if target.exists(): | |
| shutil.rmtree(target) | |
| shutil.copytree(source, target) | |
| def stage_adapter(source: Path, target: Path, kind: str) -> Path: | |
| if not (source / "adapter_config.json").is_file(): | |
| raise FileNotFoundError(source / "adapter_config.json") | |
| if not (source / "adapter_model.safetensors").is_file(): | |
| raise FileNotFoundError(source / "adapter_model.safetensors") | |
| if target.exists(): | |
| shutil.rmtree(target) | |
| target.mkdir(parents=True, exist_ok=True) | |
| for name in ( | |
| "adapter_config.json", | |
| "adapter_model.safetensors", | |
| "README.md", | |
| "tokenizer.json", | |
| "tokenizer.model", | |
| "tokenizer_config.json", | |
| "special_tokens_map.json", | |
| "chat_template.jinja", | |
| ): | |
| path = source / name | |
| if path.is_file(): | |
| copy_required_file(path, target / name) | |
| for name in ("prepare_input.py", "re_sft_common.py"): | |
| copy_required_file(REPO / name, target / name) | |
| copytree_replace(REPO / "resources", target / "resources") | |
| trainer_state_path = source / "trainer_state.json" | |
| trainer_state: Dict[str, Any] = {} | |
| if trainer_state_path.is_file(): | |
| trainer_state = json.loads(trainer_state_path.read_text(encoding="utf-8")) | |
| inference_config = { | |
| "base_model_id": BASE_MODEL_ID, | |
| "base_model_revision": BASE_MODEL_REVISION, | |
| "prompt_version": PROMPT_VERSION, | |
| "objective": "row_local_option_choice_cross_entropy", | |
| "loss_mode": "row_local_choice", | |
| "calibration_performed": False, | |
| "calibrate_no_relation": False, | |
| "no_relation_logit_bias": 0.0, | |
| "raw_logits_equal_final_logits": True, | |
| "candidate_level_decision_rule": "row_local_softmax_no_relation_threshold_0.5_else_best_positive", | |
| "no_relation_threshold": 0.5, | |
| "adapter_kind": kind, | |
| "adapter_source": str(source), | |
| "global_step": trainer_state.get("global_step"), | |
| "best_metric": trainer_state.get("best_metric"), | |
| "best_model_checkpoint": trainer_state.get("best_model_checkpoint"), | |
| } | |
| (target / "inference_config.json").write_text( | |
| json.dumps(inference_config, ensure_ascii=False, indent=2) + "\n", | |
| encoding="utf-8", | |
| ) | |
| (target / ".dru_adapter_revision").write_text( | |
| f"local-{kind}-step-{trainer_state.get('global_step', 'unknown')}\n", | |
| encoding="utf-8", | |
| ) | |
| return target | |
| def load_runner(): | |
| if not REFERENCE_RUNNER.is_file(): | |
| raise FileNotFoundError(REFERENCE_RUNNER) | |
| runner_dir = REFERENCE_RUNNER.parent | |
| sys.path.insert(0, str(runner_dir)) | |
| spec = importlib.util.spec_from_file_location( | |
| "dru_yehia2_threshold_0_5_runner", REFERENCE_RUNNER | |
| ) | |
| if spec is None or spec.loader is None: | |
| raise RuntimeError(f"Could not load runner: {REFERENCE_RUNNER}") | |
| module = importlib.util.module_from_spec(spec) | |
| sys.modules[spec.name] = module | |
| spec.loader.exec_module(module) | |
| return module | |
| def run_one(kind: str, adapter_dir: Path, output_dir: Path, input_path: Path) -> None: | |
| if output_dir.exists(): | |
| shutil.rmtree(output_dir) | |
| output_dir.mkdir(parents=True, exist_ok=True) | |
| os.environ["TEST_JSONL"] = str(input_path) | |
| os.environ["OUTPUT_DIR"] = str(output_dir) | |
| os.environ["AGGREGATION_METHOD"] = "majority_vote" | |
| os.environ["WRITE_ALTERNATIVE_SUBMISSIONS"] = "true" | |
| os.environ["YEHIA_ADAPTER_MODEL_ID"] = f"local/{kind}" | |
| os.environ["YEHIA_ADAPTER_REVISION"] = "local" | |
| os.environ["LOCAL_YEHIA_MODEL_DIR"] = str(WORKSPACE / "models" / "Yehia-7B-preview") | |
| os.environ["VLLM_USE_V1"] = "0" | |
| os.environ.setdefault("VLLM_GPU_MEMORY_UTILIZATION", "0.90") | |
| os.environ.setdefault("VLLM_ENABLE_PREFIX_CACHING", "true") | |
| os.environ.setdefault("VLLM_REQUEST_BATCH_SIZE", "512") | |
| os.environ.setdefault("TYPE_BATCH_SIZE", "64") | |
| os.environ.setdefault("YEHIA_BATCH_SIZE", "16") | |
| os.environ["FINAL_SUBMISSION_PATH"] = str(output_dir / "submission.zip") | |
| runner = load_runner() | |
| def resolve_adapter_snapshot(_: Path) -> Tuple[Path, str]: | |
| revision = (adapter_dir / ".dru_adapter_revision").read_text(encoding="utf-8").strip() | |
| (output_dir / "adapter_revision.txt").write_text(revision + "\n", encoding="utf-8") | |
| return adapter_dir, revision | |
| original_selected_inference_config = runner.selected_inference_config | |
| def selected_inference_config(path: Path) -> Dict[str, Any]: | |
| config = original_selected_inference_config(path) | |
| bias = float(config["no_relation_logit_bias"]) | |
| if bias != 0.0: | |
| raise RuntimeError(f"Forbidden no_relation_logit_bias={bias} in {path}") | |
| config["calibration_performed"] = False | |
| config["calibrate_no_relation"] = False | |
| return config | |
| runner.resolve_adapter_snapshot = resolve_adapter_snapshot | |
| runner.selected_inference_config = selected_inference_config | |
| old_argv = sys.argv | |
| try: | |
| sys.argv = [ | |
| str(REFERENCE_RUNNER), | |
| "--input", | |
| str(input_path), | |
| "--output-dir", | |
| str(output_dir), | |
| "--aggregation", | |
| "majority_vote", | |
| ] | |
| runner.main() | |
| finally: | |
| sys.argv = old_argv | |
| def main() -> None: | |
| parser = argparse.ArgumentParser() | |
| parser.add_argument("--input", type=Path, default=WORKSPACE / "test.jsonl") | |
| parser.add_argument("--which", choices=("best", "last", "all"), default="all") | |
| args = parser.parse_args() | |
| force_workspace_caches() | |
| load_dotenv(WORKSPACE / ".env", override=False) | |
| input_path = args.input.resolve() | |
| if not input_path.is_file(): | |
| raise FileNotFoundError(input_path) | |
| OUTPUT_ROOT.mkdir(parents=True, exist_ok=True) | |
| copy_required_file(input_path, OUTPUT_ROOT / "test.jsonl") | |
| adapter_sources = { | |
| "best": REPO / "best_checkpoint", | |
| "last": REPO / "latest_checkpoint", | |
| } | |
| selected = ("best", "last") if args.which == "all" else (args.which,) | |
| manifest: Dict[str, Any] = { | |
| "input": str(input_path), | |
| "output_root": str(OUTPUT_ROOT), | |
| "runner": str(REFERENCE_RUNNER), | |
| "candidate_level_decision_rule": "row_local_softmax_no_relation_threshold_0.5_else_best_positive", | |
| "no_relation_threshold": 0.5, | |
| "no_relation_logit_bias": 0.0, | |
| "methods": list(METHODS), | |
| "checkpoints": {}, | |
| } | |
| for kind in selected: | |
| staged = stage_adapter(adapter_sources[kind], OUTPUT_ROOT / "_local_adapters" / kind, kind) | |
| output_dir = OUTPUT_ROOT / kind | |
| run_one(kind, staged, output_dir, input_path) | |
| manifest["checkpoints"][kind] = { | |
| "adapter_source": str(adapter_sources[kind]), | |
| "staged_adapter": str(staged), | |
| "output_dir": str(output_dir), | |
| } | |
| (OUTPUT_ROOT / "generation_manifest.json").write_text( | |
| json.dumps(manifest, ensure_ascii=False, indent=2) + "\n", | |
| encoding="utf-8", | |
| ) | |
| if __name__ == "__main__": | |
| main() | |