File size: 5,155 Bytes
9ab951d
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
/**
 * Vietnamese number/date/text normalization for DISPLAY only.
 *
 * IMPORTANT: This only modifies the SUBTITLES and CHAT display text.
 * It does NOT affect the TTS audio output path — audio comes directly
 * from the S2S backend as PCM16, unaffected by this text processing.
 *
 * "Mất âm thanh TTS" root cause analysis:
 * The problem was NOT this file — the Vietnamese formatting examples in
 * app.js instructions were too verbose, causing session.update WebSocket
 * messages to be rejected or delayed. Fixed by simplifying instructions.
 *
 * Normalize cautiously: only run on complete (non-partial) transcripts,
 * never on every delta event. Heavy regex on 50+ deltas/second lags the
 * event loop and can starve the audio buffer processing.
 */

// ── Vietnamese number words ───────────────────────────────────────────────
const DIGITS = [
  "không", "một", "hai", "ba", "bốn", "năm", "sáu", "bảy", "tám", "chín",
];

function numberToVietnamese(n) {
  if (n === 0) return "không";
  if (n < 0) return "âm " + numberToVietnamese(-n);

  const units = ["", "nghìn", "triệu", "tỷ"];
  const groups = [];
  let temp = n;
  while (temp > 0) {
    groups.push(temp % 1000);
    temp = Math.floor(temp / 1000);
  }
  if (groups.length === 0) groups.push(0);

  const readGroup = (g) => {
    if (g === 0) return "";
    const h = Math.floor(g / 100);
    const r = g % 100;
    let s = "";
    if (h > 0) s += DIGITS[h] + " trăm ";
    else if (groups.length > 1) s += "không trăm ";
    if (r === 0) return s.trim();
    if (r < 10) {
      s += (h > 0 && r === 5) ? "lẻ năm" : "lẻ " + DIGITS[r];
    } else if (r < 20) {
      s += "mười" + (r === 10 ? "" : (r === 15 ? " lăm" : " " + DIGITS[r % 10]));
    } else {
      const t = Math.floor(r / 10);
      const o = r % 10;
      s += ["", "", "hai mươi", "ba mươi", "bốn mươi", "năm mươi",
            "sáu mươi", "bảy mươi", "tám mươi", "chín mươi"][t];
      if (o === 1) s += " mốt";
      else if (o === 5) s += " lăm";
      else if (o > 0) s += " " + DIGITS[o];
    }
    return s.replace(/\s+/g, " ").trim();
  };

  let result = "";
  for (let i = groups.length - 1; i >= 0; i--) {
    const g = groups[i];
    if (g === 0) continue;
    result += readGroup(g);
    if (i > 0) result += " " + units[i] + " ";
  }
  return result.replace(/\s+/g, " ").trim();
}

/**
 * Normalize Vietnamese text for display: convert numbers/dates to words.
 * Only handles patterns that are likely to occur in Vietnamese assistant
 * responses. Runs ONLY on final transcripts, not partial deltas.
 */
export function normalizeVietnameseText(text) {
  if (!text || typeof text !== "string") return text;

  let result = text;

  // 1. Dates: dd/mm/yyyy
  result = result.replace(
    /\b(\d{1,2})\/(\d{1,2})\/(\d{4})\b/g,
    (_, d, m, y) => `ngày ${parseInt(d)} tháng ${parseInt(m)} năm ${numberToVietnamese(parseInt(y))}`,
  );

  // 2. Percentages
  result = result.replace(/\b(\d+(?:[.,]\d+)?)%\b/g, (_, num) => {
    const val = parseFloat(num.replace(",", "."));
    return Number.isInteger(val)
      ? `${numberToVietnamese(val)} phần trăm`
      : `${numberToVietnamese(Math.floor(val))} phẩy ${numberToVietnamese(Math.round((val % 1) * 100))} phần trăm`;
  });

  // 3. Currency VND: 1.000.000₫ or 500k
  result = result.replace(/\b(\d{1,3}(?:\.\d{3})+)\s*(₫|đ|vnd)\b/gi, (_, num) =>
    `${numberToVietnamese(parseInt(num.replace(/\./g, "")))} đồng`);
  result = result.replace(/\b(\d+)k\b/gi, (_, num) =>
    `${numberToVietnamese(parseInt(num) * 1000)} đồng`);

  // 4. Decimals (not years — 1900-2100 are already handled by the model)
  result = result.replace(/\b(\d+)\.(\d{1,3})\b/g, (_, intPart, decPart) => {
    if (intPart.length > 4) return `${intPart}.${decPart}`; // not a decimal
    const decVal = parseInt(decPart);
    return `${numberToVietnamese(parseInt(intPart))} phẩy ${numberToVietnamese(decVal)}`;
  });

  // 5. Large standalone numbers (3+ digits = years, prices, quantities)
  // Only replace if surrounded by Vietnamese text context
  result = result.replace(/\b(\d{3,})\b/g, (match) => {
    const n = parseInt(match);
    if (n >= 1900 && n <= 2100) return `năm ${numberToVietnamese(n)}`;
    return numberToVietnamese(n);
  });

  // Clean spaces
  result = result.replace(/\s+/g, " ").trim();
  return result;
}

/**
 * Check if text contains Vietnamese characters.
 */
export function containsVietnamese(text) {
  return /[àáạảãâầấậẩẫăằắặẳẵèéẹẻẽêềếệểễìíịỉĩòóọỏõôồốộổỗơờớợởỡùúụủũưừứựửữỳýỵỷỹđ]/i.test(text);
}

/**
 * Smart normalizer: only run on Vietnamese text, only on final transcripts.
 * Pass `partial=false` from the transcript event to skip partial deltas.
 */
export function smartNormalize(text, partial = false) {
  if (!text || partial) return text;
  if (containsVietnamese(text)) {
    return normalizeVietnameseText(text);
  }
  return text;
}