Text Generation
Transformers
Safetensors
English
nemotron_h
ai-text-detection
idea-provenance
conversational
Instructions to use rishanthrajendhran/IdeaLens-NoParaphrase with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use rishanthrajendhran/IdeaLens-NoParaphrase with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="rishanthrajendhran/IdeaLens-NoParaphrase") messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# pip install -U transformers accelerate # Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("rishanthrajendhran/IdeaLens-NoParaphrase") model = AutoModelForCausalLM.from_pretrained("rishanthrajendhran/IdeaLens-NoParaphrase", device_map="auto") messages = [ {"role": "user", "content": "Who are you?"}, ] inputs = tokenizer.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=256) print(tokenizer.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use rishanthrajendhran/IdeaLens-NoParaphrase with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "rishanthrajendhran/IdeaLens-NoParaphrase" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "rishanthrajendhran/IdeaLens-NoParaphrase", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/rishanthrajendhran/IdeaLens-NoParaphrase
- SGLang
How to use rishanthrajendhran/IdeaLens-NoParaphrase with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "rishanthrajendhran/IdeaLens-NoParaphrase" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "rishanthrajendhran/IdeaLens-NoParaphrase", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "rishanthrajendhran/IdeaLens-NoParaphrase" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "rishanthrajendhran/IdeaLens-NoParaphrase", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use rishanthrajendhran/IdeaLens-NoParaphrase with Docker Model Runner:
docker model run hf.co/rishanthrajendhran/IdeaLens-NoParaphrase
File size: 5,092 Bytes
3525747 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 | """Apply the LoRA adapter in this repo's adapter/ folder to the base model, then score role-labelled outlines.
adapter/ holds the adapter as trained, in the layout of the Tinker training service. Do not load it with
peft.PeftModel. In transformers, Nemotron fuses the Mamba gate and x projections into one `in_proj`, and stores each
MoE layer's 128 routed experts as a single 3D tensor. PEFT has no module to attach those LoRA weights to and skips them
without a warning. This script instead merges every LoRA delta, W += (alpha / r) * B @ A, into the base weights in
place, following the same rules as the tinker-cookbook merge (tinker_cookbook.weights.build_hf_model). The resulting
weights are bit-identical to the merged model in this repo.
Needs transformers >= 5.15, torch, safetensors and huggingface_hub. The weights take 59 GiB of GPU memory and each
input adds about 4.2 MiB per token, so one 80 GB GPU handles inputs up to about 4,000 tokens; see the model card.
"""
import json
import torch
from huggingface_hub import snapshot_download
from safetensors.torch import load_file
from transformers import AutoModelForCausalLM, AutoTokenizer
REPO = "rishanthrajendhran/IdeaLens-NoParaphrase"
BASE = "nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-BF16"
SYSTEM = 'Given a role-labelled outline of a document, answer with one word: human if the source document was human-written, ai if it was AI-generated.'
SUFFIX = "<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\n"
HUMAN, AI = 50755, 2464
CUT = 0.09947 # P(human) at or below this is flagged: the global cut at 1% FPR (thresholds.json)
def apply_tinker_lora(model, adapter_dir):
"""Merge a Tinker LoRA archive for Nemotron-3.5 into `model` in place. Returns the parameter names changed."""
cfg = json.load(open(f"{adapter_dir}/adapter_config.json"))
scale = cfg["lora_alpha"] / cfg["r"]
lora = load_file(f"{adapter_dir}/adapter_model.safetensors")
params = dict(model.named_parameters())
touched = []
for key_a in sorted(k for k in lora if k.endswith(".lora_A.weight")):
A, B = lora[key_a], lora[key_a.replace(".lora_A.", ".lora_B.")]
if A.numel() == 0: # the experts have no gate projection; Tinker keeps an empty w3 placeholder
continue
name = key_a.removeprefix("base_model.model.").removesuffix(".lora_A.weight")
rows = None
if name == "model.lm_head": # Tinker nests the LM head under model.
target = "lm_head.weight"
elif name.endswith((".gate_proj", ".x_proj")): # Mamba: in_proj rows are [gate | x | B | C | dt]
layer, proj = name.rsplit(".", 1)
target = f"{layer}.in_proj.weight"
start = 0 if proj == "gate_proj" else lora[f"base_model.model.{layer}.gate_proj.lora_B.weight"].shape[0]
rows = slice(start, start + B.shape[0])
elif name.endswith(".experts.w1"): # routed experts: w1 = up_proj, one (expert, out, in) tensor per layer
target = name.removesuffix("w1") + "up_proj"
elif name.endswith(".experts.w2"): # w2 = down_proj
target = name.removesuffix("w2") + "down_proj"
else: # attention, Mamba out_proj, shared experts
target = name + ".weight"
W = params[target] # KeyError here means the adapter does not match this model
# For experts one side is shared (leading dim 1) and broadcasts across the 128 experts.
delta = scale * torch.matmul(B.to(W.device, torch.float32), A.to(W.device, torch.float32))
with torch.no_grad():
view = W.data if rows is None else W.data[rows]
assert view.shape == delta.shape, (target, tuple(view.shape), tuple(delta.shape))
view.copy_((view.float() + delta).to(W.dtype))
touched.append(target)
return touched
def load_model(repo=REPO, device_map="auto"):
tok = AutoTokenizer.from_pretrained(BASE)
model = AutoModelForCausalLM.from_pretrained(BASE, dtype=torch.bfloat16, device_map=device_map).eval()
adapter_dir = snapshot_download(repo, allow_patterns=["adapter/*"]) + "/adapter"
apply_tinker_lora(model, adapter_dir)
return model, tok
@torch.no_grad()
def p_human(model, tok, outline):
"""P(human) for one outline: one `[Role] content` line per item."""
ids = tok.encode(f"<|im_start|>system\n{SYSTEM}<|im_end|>\n<|im_start|>user\n{outline}{SUFFIX}",
add_special_tokens=False)
logits = model(torch.tensor([ids], device=model.device)).logits[0, -1].float()
return torch.softmax(logits[[HUMAN, AI]], -1)[0].item()
if __name__ == "__main__":
model, tok = load_model()
outline = ("[Central Development] A town's water supply fails after a drought, and residents organise to share wells.\n"
"[Background Context] The reservoir has been shrinking for three summers.\n"
"[Open Question] Whether the council will fund a new pipeline remains undecided.")
p = p_human(model, tok, outline)
print(f"P(human) = {p:.4f} -> {'flagged as AI ideas' if p <= CUT else 'not flagged'}")
|