Spaces:
Sleeping
Sleeping
File size: 1,268 Bytes
f9a6bac | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 | import torch
class ResumeParser:
def __init__(self, model, tokenizer):
self.model = model
self.tokenizer = tokenizer
def format_data(self, txt):
prompt = f"""<|im_start|>system
Extract and format resume: name | role | years | skills | core skills | experiences | education | certifications.
Extract ONLY from input. NO inference or additions.
<|im_end|>
<|im_start|>user
{txt}
<|im_end|>
<|im_start|>assistant
"""
return prompt
def parse(self, resume):
inputs = self.tokenizer(resume, return_tensors="pt", padding=False, truncation=True)
with torch.no_grad():
# Mixed precision for faster CPU inference
with torch.cpu.amp.autocast():
output = self.model.generate(
**inputs,
max_new_tokens=300,
do_sample=False,
use_cache=True # KV cache speeds up token generation
)
decoded = self.tokenizer.decode(output[0], skip_special_tokens=False)
decoded = decoded.split("<|im_start|>assistant\n")[1].split("\n<|im_end|>")[0]
return decoded
|