Spaces:
Sleeping
Sleeping
Download modules/resume_parser.py from deepsu/HR_TOOL_V2: direct link, hf CLI and curl.
- Browser
- Download file 1.27 kB
-
https://huggingface.co/spaces/deepsu/HR_TOOL_V2/resolve/main/modules/resume_parser.py
- Command line
-
hf download hf://spaces/deepsu/HR_TOOL_V2/modules/resume_parser.py
-
curl -L -o resume_parser.py https://huggingface.co/spaces/deepsu/HR_TOOL_V2/resolve/main/modules/resume_parser.py
1.27 kB
| import torch | |
| class ResumeParser: | |
| def __init__(self, model, tokenizer): | |
| self.model = model | |
| self.tokenizer = tokenizer | |
| def format_data(self, txt): | |
| prompt = f"""<|im_start|>system | |
| Extract and format resume: name | role | years | skills | core skills | experiences | education | certifications. | |
| Extract ONLY from input. NO inference or additions. | |
| <|im_end|> | |
| <|im_start|>user | |
| {txt} | |
| <|im_end|> | |
| <|im_start|>assistant | |
| """ | |
| return prompt | |
| def parse(self, resume): | |
| inputs = self.tokenizer(resume, return_tensors="pt", padding=False, truncation=True) | |
| with torch.no_grad(): | |
| # Mixed precision for faster CPU inference | |
| with torch.cpu.amp.autocast(): | |
| output = self.model.generate( | |
| **inputs, | |
| max_new_tokens=300, | |
| do_sample=False, | |
| use_cache=True # KV cache speeds up token generation | |
| ) | |
| decoded = self.tokenizer.decode(output[0], skip_special_tokens=False) | |
| decoded = decoded.split("<|im_start|>assistant\n")[1].split("\n<|im_end|>")[0] | |
| return decoded | |