"""Serving-time post-processing for the frozen persona (35B main_v5 + prompt v0). Rule from the owner: on every second reply, the crying emoji, which the model uses as sentence punctuation, becomes a period. Usage: from postprocess import Postprocessor; pp = Postprocessor(); text = pp(text)""" import re EMO = '😭' def emoji_to_period(text: str) -> str: # 😭 followed by end/newline/space+capital/space+lowercase: it closes a sentence -> "." # 😭 directly before existing punctuation: just drop it. Preceding space is absorbed. text = re.sub(r'\s*😭+(?=\s*[.!?,])', '', text) # "...money 😭." -> "...money." text = re.sub(r'\s*😭+(?=\s*$|\s*\n)', '.', text) # end of text or line -> "." text = re.sub(r'\s*😭+\s+(?=\S)', '. ', text) # mid-text -> ". next" text = re.sub(r'\.\s*\.', '.', text) # collapse doubles # capitalize the word after a period we inserted, when the model wrote it lowercase mid-line return re.sub(r'(\. )([a-z])', lambda m: m.group(1) + m.group(2).upper(), text) class Postprocessor: def __init__(self): self.n = 0 def __call__(self, text: str) -> str: self.n += 1 return emoji_to_period(text) if self.n % 2 == 0 else text if __name__ == '__main__': import json, sys rows = [json.loads(l) for l in open(sys.argv[1])] ctx = [r['reply'] for r in rows if r['kind'] == 'ctx' and r.get('expect') == 'slang_on' and EMO in r['reply']] for t in ctx[:4]: print('BEFORE:', t.strip()[:230], '\nAFTER: ', emoji_to_period(t).strip()[:230], '\n')