File size: 45,329 Bytes
ee888e1
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
// Phase 3b: LLM enrichment for the WorldMonitor Brief envelope.
//
// Substitutes the stubbed `whyMatters` per story and the stubbed
// executive summary (`digest.lead` / `digest.threads` / `digest.signals`)
// with Gemini 2.5 Flash output via the existing OpenRouter-backed
// callLLM chain. The LLM provider is pinned to openrouter by
// skipProviders:['ollama','groq'] so the brief's editorial voice
// stays on one model across environments.
//
// Deliberately:
//   - Pure parse/build helpers are exported for testing without IO.
//   - Cache layer is parameterised (cacheGet / cacheSet) so tests use
//     an in-memory stub and production uses Upstash.
//   - Any failure (null LLM result, parse error, cache hiccup) falls
//     through to the original stub β€” the brief must always ship.
//
// Cache semantics:
//   - brief:llm:whymatters:v6:{storyHash} β€” 24h, shared across users
//     for the same story. v4 bumped from v3 alongside the F6
//     date-grounding line: every v3 row was produced from a prompt
//     with no notion of "today" and may state a fabricated year, so
//     v3 rows must not survive the deploy. v2 rows were lead-blind.
//   - brief:llm:digest:v8:{userId|public}:{sensitivity}:{poolHash}
//     β€” 4h. The canonical synthesis is now ALWAYS produced through
//     this path (formerly split with `generateAISummary` in the
//     digest cron). Material includes profile-SHA, greeting bucket,
//     isPublic flag, and per-story hash so cache hits never serve a
//     differently-ranked or differently-personalised prompt.
//     When isPublic=true, the userId slot in the key is the literal
//     string 'public' so all public-share readers of the same
//     (date, sensitivity, story-pool) hit the same row β€” no PII in
//     the public cache key. v6 bumped from v5 for the F6
//     date-grounding line (same reason as whymatters v4); v5 landed
//     the grounding validator after the May 12 hallucination β€” see
//     generateDigestProse header comment.

import { createHash } from 'node:crypto';

import {
  WHY_MATTERS_SYSTEM,
  WHY_MATTERS_V1_MAX_CHARS,
  WHY_MATTERS_V1_MIN_CHARS,
  WHY_MATTERS_V2_MAX_CHARS,
  WHY_MATTERS_V2_MIN_CHARS,
  briefDateLine,
  buildWhyMattersUserPrompt,
  hashBriefStory,
  hasTerminalPunctuation,
  parseWhyMatters,
  checkLeadGrounding,
  leadGroundsAgainstStory,
} from '../../shared/brief-llm-core.js';

// #4921: the grounding spine now lives in shared/brief-llm-core.js β€” re-export
// for existing consumers of this module.
export { checkLeadGrounding, leadGroundsAgainstStory };
import { sanitizeForPrompt } from '../../server/_shared/llm-sanitize.js';
// Single source of truth for the brief story cap. Both buildDigestPrompt
// and hashDigestInput must slice to this value or the LLM prose drifts
// from the rendered story cards (PR #3389 reviewer P1).
import { MAX_STORIES_PER_USER } from './brief-compose.mjs';

/**
 * Sanitize the story fields that flow into buildWhyMattersUserPrompt and
 * buildStoryDescriptionPrompt. Mirrors
 * server/worldmonitor/intelligence/v1/brief-why-matters-prompt.ts
 * sanitizeStoryFields β€” the legacy Railway fallback path must apply the
 * same defense as the analyst endpoint, since this is exactly what runs
 * when the endpoint misses / returns null / throws.
 *
 * `description` is included because the RSS-description fix (2026-04-24)
 * now threads untrusted article bodies into the description prompt as
 * grounding context. Without sanitising it, a hostile feed's
 * `<description>` is an unsanitised injection vector β€” the asymmetry with
 * whyMatters (already sanitised) was a latent bug, fixed here.
 *
 * Kept local (not promoted to brief-llm-core.js) because llm-sanitize.js
 * only lives in server/_shared and the edge endpoint already sanitizes
 * before its own buildWhyMattersUserPrompt call.
 *
 * @param {{ headline?: string; source?: string; threatLevel?: string; category?: string; country?: string; description?: string }} story
 */
function sanitizeStoryForPrompt(story) {
  return {
    headline: sanitizeForPrompt(story.headline ?? ''),
    source: sanitizeForPrompt(story.source ?? ''),
    threatLevel: sanitizeForPrompt(story.threatLevel ?? ''),
    category: sanitizeForPrompt(story.category ?? ''),
    country: sanitizeForPrompt(story.country ?? ''),
    description: sanitizeForPrompt(story.description ?? ''),
  };
}

// Re-export for backcompat with existing tests / callers.
export { WHY_MATTERS_SYSTEM, hashBriefStory, parseWhyMatters };
export const buildWhyMattersPrompt = buildWhyMattersUserPrompt;

// ── Tunables ───────────────────────────────────────────────────────────────

const WHY_MATTERS_TTL_SEC = 24 * 60 * 60;
const DIGEST_PROSE_TTL_SEC = 4 * 60 * 60;
const STORY_DESCRIPTION_TTL_SEC = 24 * 60 * 60;
const WHY_MATTERS_CONCURRENCY = 5;

// Pin to openrouter (google/gemini-2.5-flash until the #4944 U4 brief-voice
// cutover, which is gated on the U3 shadow evaluation). Ollama isn't deployed
// in Railway, and pinning keeps the brief's editorial voice on one model
// across environments instead of drifting to the groq fallback.
const BRIEF_LLM_SKIP_PROVIDERS = ['ollama', 'groq'];

// ── whyMatters (per story) ─────────────────────────────────────────────────
// The pure helpers (`WHY_MATTERS_SYSTEM`, `buildWhyMattersUserPrompt` (aliased
// to `buildWhyMattersPrompt` for backcompat), `parseWhyMatters`, `hashBriefStory`)
// live in `shared/brief-llm-core.js` so the Vercel-edge endpoint
// (`api/internal/brief-why-matters.ts`) can import them without pulling in
// `node:crypto`. See the `shared/` β†’ `scripts/shared/` mirror convention.

function normalizeAnalystWhyMatters(value) {
  if (typeof value !== 'string') return null;
  const normalized = value.trim();
  const minChars = Math.min(WHY_MATTERS_V1_MIN_CHARS, WHY_MATTERS_V2_MIN_CHARS);
  const maxChars = Math.max(WHY_MATTERS_V1_MAX_CHARS, WHY_MATTERS_V2_MAX_CHARS);
  if (normalized.length < minChars || normalized.length > maxChars) return null;
  if (/^story flagged by your sensitivity/i.test(normalized)) return null;
  return hasTerminalPunctuation(normalized) ? normalized : null;
}

/**
 * Resolve a `whyMatters` sentence for one story.
 *
 * Four-layer graceful degradation:
 *   1. `deps.callAnalystWhyMatters(story)` β€” the analyst-context edge
 *      endpoint (brief:llm:whymatters:v10 cache lives there). Preferred.
 *   2. Direct read of the endpoint's v10 envelope cache (#4914) β€” the
 *      endpoint CALL can fail while its cached envelope is still valid;
 *      reusing it avoids a paid duplicate generation.
 *   3. Legacy direct-Gemini chain: cacheGet (v6) β†’ callLLM β†’ cacheSet.
 *      Runs whenever the analyst call is missing, returns null, or throws.
 *   4. Caller (enrichBriefEnvelopeWithLLM) uses the baseline stub if
 *      this function returns null.
 *
 * Returns null on all-layer failure.
 *
 * @param {object} story
 * @param {{
 *   callLLM: (system: string, user: string, opts: object) => Promise<string|null>;
 *   cacheGet: (key: string) => Promise<unknown>;
 *   cacheSet: (key: string, value: unknown, ttlSec: number) => Promise<void>;
 *   callAnalystWhyMatters?: (story: object) => Promise<string|null>;
 * }} deps
 */
export async function generateWhyMatters(story, deps) {
  // Priority path: analyst endpoint. It owns its own cache and has
  // ALREADY validated the output via parseWhyMatters (gemini path) or
  // parseWhyMattersV2 (analyst path, multi-sentence). We must NOT
  // re-parse here with the narrower v1 parser β€” v2 intentionally permits
  // longer multi-sentence output. Trust the wire shape; only reject an
  // obviously-bad payload (empty, stub
  // echo, incomplete sentence, or length outside either parser's bounds).
  if (typeof deps.callAnalystWhyMatters === 'function') {
    try {
      const analystOut = await deps.callAnalystWhyMatters(story);
      const normalized = normalizeAnalystWhyMatters(analystOut);
      if (normalized) return normalized;
      if (typeof analystOut === 'string') {
        console.warn(
          `[brief-llm] callAnalystWhyMatters β†’ fallback: endpoint returned out-of-bounds, stub, or incomplete prose (len=${analystOut.trim().length})`,
        );
      } else {
        const responseType = analystOut === null ? 'null' : typeof analystOut;
        console.warn(
          `[brief-llm] callAnalystWhyMatters β†’ fallback: endpoint returned no usable string (type=${responseType})`,
        );
      }
    } catch (err) {
      console.warn(
        `[brief-llm] callAnalystWhyMatters β†’ fallback: ${err instanceof Error ? err.message : String(err)}`,
      );
    }
  }

  // #4914: before paying a direct-Gemini generation, check the analyst
  // endpoint's OWN cache namespace. api/internal/brief-why-matters.ts
  // stores its envelope at brief:llm:whymatters:v10:{hash} under the same
  // hashBriefStory identity β€” when the endpoint CALL failed transiently
  // (or no endpoint is configured), the story may already have a paid,
  // validated envelope sitting in Redis. Read-only: this fallback's own
  // fallback output stays in the legacy v6 namespace below, so the
  // two prompt contracts never cross-contaminate in the write direction.
  const storyHash = await hashBriefStory(story);
  try {
    const v10 = await deps.cacheGet(`brief:llm:whymatters:v10:${storyHash}`);
    if (v10 && typeof v10 === 'object') {
      const normalized = normalizeAnalystWhyMatters(v10.whyMatters);
      if (normalized) return normalized;
    }
  } catch { /* treat as miss */ }

  // Fallback path: legacy direct-Gemini chain with the v4 cache.
  // Bumped v3β†’v4 on 2026-05-14 alongside the F6 date-grounding line:
  // every v3 row was produced from a buildWhyMattersPrompt prompt with
  // no notion of "today", so a v3 row may state a fabricated year
  // (the bug F6 fixes). Serving v3 on a cache hit would keep shipping
  // that fabrication for the 24h TTL β€” the prefix bump forces a clean
  // cold-start through the date-grounded prompt on first tick after
  // deploy. (v2β†’v3 was the 2026-04-24 RSS-description fix.) Entries
  // expire in ≀24h so the prior prefix ages out without a DEL sweep.
  //
  // v4β†’v5: 2026-05-17 PR #3751. `hashBriefStory` folds `story.category`
  // into the key; pre-PR every story carried 'General' (no category was
  // persisted on story:track:v1), post-PR carries the per-story
  // Title-Cased EventCategory value. Every v4 cache row is now stale.
  // Bump invalidates them cleanly.
  //
  // v5β†’v6: 2026-07-10 issue #5168. v5 rows were written before the Railway
  // provider chain rejected finish_reason=length, so an abbreviation-ending
  // token clip could be cached as an apparently complete sentence. The old
  // rows carry no completion metadata and cannot be distinguished safely.
  const key = `brief:llm:whymatters:v6:${storyHash}`;
  try {
    const hit = await deps.cacheGet(key);
    const parsedHit = parseWhyMatters(hit);
    if (parsedHit) return parsedHit;
  } catch { /* cache miss is fine */ }
  // Sanitize story fields before interpolating into the prompt. The analyst
  // endpoint already does this; without it the Railway fallback path was an
  // unsanitized injection vector for any future untrusted `source` / `headline`.
  const { system, user } = buildWhyMattersPrompt(sanitizeStoryForPrompt(story));
  let text = null;
  try {
    text = await deps.callLLM(system, user, {
      maxTokens: 120,
      temperature: 0.4,
      timeoutMs: 10_000,
      skipProviders: BRIEF_LLM_SKIP_PROVIDERS,
      stage: 'brief-whymatters-cron',
    });
  } catch {
    return null;
  }
  const parsed = parseWhyMatters(text);
  if (!parsed) return null;
  try {
    await deps.cacheSet(key, parsed, WHY_MATTERS_TTL_SEC);
  } catch { /* cache write failures don't matter here */ }
  return parsed;
}

// ── Per-story description (replaces title-verbatim fallback) ──────────────

const STORY_DESCRIPTION_SYSTEM =
  'You are the editor of WorldMonitor Brief, a geopolitical intelligence magazine. ' +
  'Given the story attributes below, write ONE concise sentence (16–30 words) that ' +
  'describes the development itself β€” not why it matters, not the reader reaction. ' +
  'Editorial, serious, past/present tense, named actors where possible. Do NOT ' +
  'repeat the headline verbatim. No preamble, no quotes, no questions, no markdown, ' +
  'no hedging. One sentence only.';

/**
 * @param {{ headline: string; source: string; category: string; country: string; threatLevel: string; description?: string }} story
 * @returns {{ system: string; user: string }}
 */
export function buildStoryDescriptionPrompt(story) {
  // Grounding context: when the RSS feed carried a real description
  // (post-RSS-description fix, 2026-04-24), interpolate it as `Context:`
  // between the metadata block and the "One editorial sentence" instruction.
  // This is the actual fix for the named-actor hallucination class β€” the LLM
  // now has the article's body to paraphrase instead of filling role-label
  // headlines from its parametric priors. Skip when description is empty or
  // normalise-equal to the headline (no grounding value; parser already
  // filters this but the prompt builder is a second belt-and-braces check).
  const normalise = /** @param {string} x */ (x) => x.trim().toLowerCase().replace(/\s+/g, ' ');
  const rawDescription = typeof story.description === 'string' ? story.description.trim() : '';
  const contextUseful = rawDescription.length > 0
    && normalise(rawDescription) !== normalise(story.headline ?? '');
  const contextLine = contextUseful ? `Context: ${rawDescription.slice(0, 400)}` : null;

  const lines = [
    `Headline: ${story.headline}`,
    `Source: ${story.source}`,
    `Severity: ${story.threatLevel}`,
    `Category: ${story.category}`,
    `Country: ${story.country}`,
    ...(contextLine ? [contextLine] : []),
    '',
    'One editorial sentence describing what happened (not why it matters):',
  ];
  return { system: STORY_DESCRIPTION_SYSTEM, user: lines.join('\n') };
}

/**
 * Parse + validate the LLM story-description output. Rejects empty
 * responses, boilerplate preambles that slipped through the system
 * prompt, outputs that trivially echo the headline (sanity guard
 * against models that default to copying the prompt), and lengths
 * that drift far outside the prompted range.
 *
 * @param {unknown} text
 * @param {string} [headline]  used to detect headline-echo drift
 * @returns {string | null}
 */
export function parseStoryDescription(text, headline) {
  if (typeof text !== 'string') return null;
  let s = text.trim();
  if (!s) return null;
  s = s.replace(/^[\u201C"']+/, '').replace(/[\u201D"']+$/, '').trim();
  const match = s.match(/^[^.!?]+[.!?]/);
  const sentence = match ? match[0].trim() : s;
  if (sentence.length < 40 || sentence.length > 400) return null;
  if (typeof headline === 'string') {
    const normalise = /** @param {string} x */ (x) => x.trim().toLowerCase().replace(/\s+/g, ' ');
    // Reject outputs that are a verbatim echo of the headline β€” that
    // is exactly the fallback we're replacing, shipping it as
    // "LLM enrichment" would be dishonest about cache spend.
    if (normalise(sentence) === normalise(headline)) return null;
  }
  return sentence;
}

/**
 * Resolve a description sentence for one story via cache β†’ LLM.
 * Returns null on any failure; caller falls back to the composer's
 * baseline (cleaned headline) rather than shipping with a placeholder.
 *
 * @param {object} story
 * @param {{
 *   callLLM: (system: string, user: string, opts: object) => Promise<string|null>;
 *   cacheGet: (key: string) => Promise<unknown>;
 *   cacheSet: (key: string, value: unknown, ttlSec: number) => Promise<void>;
 * }} deps
 */
export async function generateStoryDescription(story, deps) {
  // Shares hashBriefStory() with whyMatters β€” the key prefix
  // (`brief:llm:description:v3:`) is what separates the two cache
  // namespaces; the material is the six fields including description.
  // Bumped v1β†’v2 on 2026-04-24 alongside the RSS-description fix so
  // cached pre-grounding output (hallucinated named actors from
  // headline-only prompts) is evicted. hashBriefStory itself includes
  // description in the hash material, so content drift invalidates
  // naturally too β€” the prefix bump is belt-and-braces.
  //
  // v2β†’v3: 2026-05-17 PR #3751. `hashBriefStory` folds `story.category`
  // into the hash material β€” same story-shape change as whymatters
  // v4β†’v5. Pre-PR every category was 'General'; post-PR carries the
  // per-story Title-Cased EventCategory. Bump invalidates v2 entries.
  const key = `brief:llm:description:v3:${await hashBriefStory(story)}`;
  try {
    const hit = await deps.cacheGet(key);
    if (typeof hit === 'string') {
      // Revalidate on cache hit so a pre-fix bad row (short, echo,
      // malformed) can't flow into the envelope unchecked.
      const valid = parseStoryDescription(hit, story.headline);
      if (valid) return valid;
    }
  } catch { /* cache miss is fine */ }
  // Sanitise the story BEFORE building the prompt. `description` (RSS body)
  // is untrusted input; without sanitisation, a hostile feed's
  // `<description>` would be an injection vector. The whyMatters path
  // already does this β€” keep the two symmetric.
  const { system, user } = buildStoryDescriptionPrompt(sanitizeStoryForPrompt(story));
  let text = null;
  try {
    text = await deps.callLLM(system, user, {
      maxTokens: 140,
      temperature: 0.4,
      timeoutMs: 10_000,
      skipProviders: BRIEF_LLM_SKIP_PROVIDERS,
      stage: 'brief-description-cron',
    });
  } catch {
    return null;
  }
  const parsed = parseStoryDescription(text, story.headline);
  if (!parsed) return null;
  try {
    await deps.cacheSet(key, parsed, STORY_DESCRIPTION_TTL_SEC);
  } catch { /* ignore */ }
  return parsed;
}

// ── Digest prose (canonical synthesis) ─────────────────────────────────────
//
// This is the single LLM call that produces the brief's executive summary.
// All channels (email HTML, plain-text, Telegram, Slack, Discord, webhook)
// AND the magazine's `digest.lead` read the same string from this output.
// The cron orchestration layer also produces a separate non-personalised
// `publicLead` via `generateDigestProsePublic` for the share-URL surface.

const DIGEST_PROSE_SYSTEM_BASE =
  'You are the chief editor of WorldMonitor Brief. Given a ranked list of ' +
  "today's top stories for a reader, produce EXACTLY this JSON and nothing " +
  'else (no markdown, no code fences, no preamble):\n' +
  '{\n' +
  '  "lead": "<2–3 sentences. The FIRST sentence MUST name the single most ' +
  "impactful development by its specific actor and event (e.g. \"Pentagon " +
  "chief Hegseth declared the US blockade on Iran is going global\"), NOT " +
  'an editorial framing about "geopolitical tensions" or "shifting ' +
  'landscapes". Subsequent sentences may give brief context about THE SAME ' +
  'story (causes, stakes, prior developments). Reference a SECOND story ONLY ' +
  'when there is a substantive link to the primary one (shared actor, causal ' +
  'connection, direct policy consequence, same geographic theatre). NEVER ' +
  'staple unrelated stories together using weak temporal connectives like ' +
  '"This comes as", "Meanwhile", "At the same time", "In other news", or ' +
  '"Elsewhere" β€” those produce editorially incoherent leads that mention two ' +
  'unrelated events in one sentence without explaining why they belong ' +
  'together. If two top stories are unrelated, just lead with the most ' +
  'impactful one and let the threads list cover the rest. No vapid hedging.>",\n' +
  '  "threads": [\n' +
  '    { "tag": "<one-word editorial category e.g. Energy, Diplomacy, Climate>", ' +
  '"teaser": "<one sentence naming a SPECIFIC event or actor β€” e.g. ' +
  '\\"Hegseth fired Navy Secretary Phelan amid Iran-policy rift\\" β€” NOT ' +
  'generic phrasing like \\"tensions continue to develop\\".>" }\n' +
  '  ],\n' +
  '  "signals": ["<forward-looking imperative phrase, <=14 words, naming a ' +
  'specific watch-item β€” e.g. \\"Watch for direct US-Iran naval engagement ' +
  'in the Strait of Hormuz\\".>"],\n' +
  '  "rankedStoryHashes": ["<short hash from the [h:XXXX] prefix of the most ' +
  'important story>", "..."]\n' +
  '}\n' +
  'BANNED phrasing (do NOT use any of these β€” they are vapid editorial ' +
  'filler that hides which events actually matter): "the global stage", ' +
  '"buzzing with developments", "intricate shifts", "evolving landscape", ' +
  '"navigating", "discerning reader", "continues to simmer", "shape the ' +
  'coming months", "strategic importance".\n' +
  'BANNED stitching phrases (do NOT use any of these to staple two stories ' +
  'together in the lead β€” they signal unrelated content awkwardly joined): ' +
  '"this comes as", "this declaration comes as", "this announcement comes as", ' +
  '"meanwhile", "at the same time", "in other news", "elsewhere", "across the ' +
  'world", "on another front", "in a separate development". If two stories ' +
  'are not substantively linked (no shared actor, no causal connection, no ' +
  'direct policy consequence, no same geographic theatre), do NOT stitch them ' +
  'into one sentence β€” lead with the more impactful one alone.\n' +
  'Threads: 3–6 items reflecting actual clusters in the stories. ' +
  'Signals: 2–4 items, forward-looking. ' +
  'rankedStoryHashes: at least the top 3 stories by editorial importance, ' +
  'using the short hash from each story line (the value inside [h:...]). ' +
  'Lead with the single most impactful development NAMED. Lead under 250 words.';

/**
 * Compute a coarse greeting bucket for cache-key stability.
 * Greeting strings can vary in punctuation/capitalisation across
 * locales; the bucket collapses them to one of three slots so the
 * cache key only changes when the time-of-day window changes.
 *
 * Unrecognised greetings (locale-specific phrases the keyword
 * heuristic doesn't match, empty strings after locale changes,
 * non-string inputs) collapse to the literal `''` slot. This is
 * INTENTIONAL β€” it's a stable fourth bucket, not a sentinel for
 * "missing data". A user whose greeting flips between a recognised
 * value (e.g. "Good morning") and an unrecognised one (e.g. a
 * locale-specific phrase) will get different cache keys, which is
 * correct: those produce visibly different leads. Greptile P2 on
 * PR #3396 raised the visibility, kept the behaviour.
 *
 * @param {string|null|undefined} greeting
 * @returns {'morning' | 'afternoon' | 'evening' | ''}
 */
export function greetingBucket(greeting) {
  if (typeof greeting !== 'string') return '';
  const g = greeting.toLowerCase();
  if (g.includes('morning')) return 'morning';
  if (g.includes('afternoon')) return 'afternoon';
  if (g.includes('evening') || g.includes('night')) return 'evening';
  return '';
}

/**
 * @typedef {object} DigestPromptCtx
 * @property {string|null} [profile]   formatted user profile lines, or null for non-personalised
 * @property {string|null} [greeting]  e.g. "Good morning", or null for non-personalised
 * @property {boolean}     [isPublic]  true = strip personalisation, build a generic lead
 * @property {string}      [todayIso]  ISO date for the date-grounding line; defaults to today (UTC)
 */

/**
 * Build the digest-prose prompt. When `ctx.profile` / `ctx.greeting`
 * are present (and `ctx.isPublic !== true`), the prompt asks the
 * model to address the reader by their watched assets/regions and
 * open with the greeting. Otherwise the prompt produces a generic
 * editorial brief safe for share-URL surfaces.
 *
 * Per-story line format includes a stable short-hash prefix:
 *   `01 [h:abc12345] [CRITICAL] Headline β€” Category Β· Country Β· Source`
 * The model emits `rankedStoryHashes` referencing those short hashes
 * so the cron can re-order envelope.stories before the cap.
 *
 * @param {Array<{ hash?: string; headline: string; threatLevel: string; category: string; country: string; source: string }>} stories
 * @param {string} sensitivity
 * @param {DigestPromptCtx} [ctx]
 * @returns {{ system: string; user: string }}
 */
export function buildDigestPrompt(stories, sensitivity, ctx = {}) {
  const isPublic = ctx?.isPublic === true;
  const profile = !isPublic && typeof ctx?.profile === 'string' ? ctx.profile.trim() : '';
  const greeting = !isPublic && typeof ctx?.greeting === 'string' ? ctx.greeting.trim() : '';

  const lines = stories.slice(0, MAX_STORIES_PER_USER).map((s, i) => {
    const n = String(i + 1).padStart(2, '0');
    const sev = (s.threatLevel ?? '').toUpperCase();
    // Short hash prefix β€” first 8 chars of digest story hash. Keeps
    // the prompt compact while remaining collision-free for ≀30
    // stories. Stories without a hash fall back to position-based
    // 'p<NN>' so the prompt is always well-formed.
    const shortHash = typeof s.hash === 'string' && s.hash.length >= 8
      ? s.hash.slice(0, 8)
      : `p${n}`;
    return `${n}. [h:${shortHash}] [${sev}] ${s.headline} β€” ${s.category} Β· ${s.country} Β· ${s.source}`;
  });

  const userParts = [
    `Reader sensitivity level: ${sensitivity}`,
  ];
  if (greeting) {
    userParts.push('', `Open the lead with: "${greeting}."`);
  }
  if (profile) {
    userParts.push('', 'Reader profile (use to personalise lead and signals):', profile);
  }
  userParts.push('', "Today's surfaced stories (ranked):", ...lines);

  // F6: the static system prompt has no notion of "now" β€” without an
  // explicit date the model fabricates years (a May 2026 brief shipped
  // a "deploy ... in 2024" line). briefDateLine pins the current date.
  return {
    system: `${DIGEST_PROSE_SYSTEM_BASE}\n${briefDateLine(ctx?.todayIso)}`,
    user: userParts.join('\n'),
  };
}

// Back-compat alias for tests that import the old constant name.
export const DIGEST_PROSE_SYSTEM = DIGEST_PROSE_SYSTEM_BASE;

/**
 * Strict shape check for a parsed digest-prose object. Used by BOTH
 * parseDigestProse (fresh LLM output) AND generateDigestProse's
 * cache-hit path, so a bad row written under an older/buggy version
 * can't poison the envelope at SETEX time. Returns a **normalised**
 * copy of the object on success, null on any shape failure β€” never
 * returns the caller's object by reference so downstream writes
 * can't observe internal state.
 *
 * v3 (2026-04-25): adds optional `rankedStoryHashes` β€” short hashes
 * (β‰₯4 chars each) that the orchestration layer maps back to digest
 * story `hash` values to re-order envelope.stories before the cap.
 * Field is optional so v2-shaped cache rows still pass validation
 * during the rollout window β€” they just don't carry ranking signal.
 *
 * v5 (2026-05-12): when `stories` is supplied, additionally runs
 * checkLeadGrounding. A shape-valid but content-fabricated lead
 * (proper nouns absent from every input headline) is rejected so
 * the caller falls through to L2/L3 instead of shipping the
 * hallucination. Back-compat: omitted/empty `stories` skips the
 * grounding check, preserving the original 1-arg behavior for
 * callers that don't have the source pool in hand.
 *
 * @param {unknown} obj
 * @param {Array<{ headline?: string }>} [stories]  source pool used to
 *   ground-check the lead. Optional for back-compat.
 * @returns {{ lead: string; threads: Array<{tag:string;teaser:string}>; signals: string[]; rankedStoryHashes: string[] } | null}
 */
export function validateDigestProseShape(obj, stories) {
  if (!obj || typeof obj !== 'object' || Array.isArray(obj)) return null;

  const lead = typeof obj.lead === 'string' ? obj.lead.trim() : '';
  if (lead.length < 40 || lead.length > 1500) return null;

  const rawThreads = Array.isArray(obj.threads) ? obj.threads : [];
  const threads = rawThreads
    .filter((t) => t && typeof t.tag === 'string' && typeof t.teaser === 'string')
    .map((t) => ({
      tag: t.tag.trim().slice(0, 40),
      teaser: t.teaser.trim().slice(0, 220),
    }))
    .filter((t) => t.tag.length > 0 && t.teaser.length > 0)
    .slice(0, 6);
  if (threads.length < 1) return null;

  // The prompt instructs the model to produce signals of "<=14 words,
  // forward-looking imperative phrase". Enforce both a word cap (with
  // a small margin of 4 words for model drift and compound phrases)
  // and a byte cap β€” a 30-word "signal" would render as a second
  // paragraph on the signals page, breaking visual rhythm. Previously
  // only the byte cap was enforced, allowing ~40-word signals to
  // sneak through when the model ignored the word count.
  const rawSignals = Array.isArray(obj.signals) ? obj.signals : [];
  const signals = rawSignals
    .filter((x) => typeof x === 'string')
    .map((x) => x.trim())
    .filter((x) => {
      if (x.length === 0 || x.length >= 220) return false;
      const words = x.split(/\s+/).filter(Boolean).length;
      return words <= 18;
    })
    .slice(0, 6);

  // rankedStoryHashes: optional. When present, must be array of
  // non-empty short-hash strings (β‰₯4 chars). Each entry trimmed and
  // capped to 16 chars (the prompt emits 8). Length capped to
  // MAX_STORIES_PER_USER Γ— 2 to bound prompt drift.
  const rawRanked = Array.isArray(obj.rankedStoryHashes) ? obj.rankedStoryHashes : [];
  const rankedStoryHashes = rawRanked
    .filter((x) => typeof x === 'string')
    .map((x) => x.trim().slice(0, 16))
    .filter((x) => x.length >= 4)
    .slice(0, MAX_STORIES_PER_USER * 2);

  // v5 grounding gate. Run AFTER shape normalisation so the
  // synthesis we evaluate is the same shape the renderer would
  // see β€” checkLeadGrounding inspects `lead` and `threads[].teaser`,
  // both already trimmed and capped above.
  if (Array.isArray(stories) && stories.length > 0
      && !checkLeadGrounding({ lead, threads }, stories, MAX_STORIES_PER_USER)) {
    return null;
  }

  return { lead, threads, signals, rankedStoryHashes };
}

/**
 * @param {unknown} text
 * @param {Array<{ headline?: string }>} [stories]  forwarded to
 *   validateDigestProseShape so fresh LLM output is grounding-checked
 *   the same way cache hits are.
 * @returns {{ lead: string; threads: Array<{tag:string;teaser:string}>; signals: string[] } | null}
 */
export function parseDigestProse(text, stories) {
  if (typeof text !== 'string') return null;
  let s = text.trim();
  if (!s) return null;
  // Defensive: strip common wrappings the model sometimes inserts
  // despite the explicit system instruction.
  s = s.replace(/^```(?:json)?\s*/i, '').replace(/\s*```$/, '').trim();
  let obj;
  try {
    obj = JSON.parse(s);
  } catch {
    return null;
  }
  return validateDigestProseShape(obj, stories);
}

/**
 * Cache key for digest prose. MUST cover every field the LLM sees,
 * in the order it sees them β€” anything less and we risk returning
 * pre-computed prose for a materially different prompt (e.g. the
 * same stories re-ranked, or with corrected category/country
 * metadata). The old "sort + headline|severity" hash was explicitly
 * about cache-hit rate; that optimisation is the wrong tradeoff for
 * an editorial product whose correctness bar is "matches the email".
 *
 * v3 key space (2026-04-25): material now includes the digest-story
 * `hash` (per-story rankability), `ctx.profile` SHA-256, greeting
 * bucket, and isPublic flag. When `ctx.isPublic === true` the userId
 * slot is replaced with the literal `'public'` so all public-share
 * readers of the same (sensitivity, story-pool) hit ONE cache row
 * regardless of caller β€” no PII in public cache keys, no per-user
 * inflation. v2 rows are ignored on rollout (paid for once).
 *
 * @param {string} userId
 * @param {Array} stories
 * @param {string} sensitivity
 * @param {DigestPromptCtx} [ctx]
 */
function hashDigestInput(userId, stories, sensitivity, ctx = {}) {
  const isPublic = ctx?.isPublic === true;
  const profileSha = isPublic ? '' : (typeof ctx?.profile === 'string' && ctx.profile.length > 0
    ? createHash('sha256').update(ctx.profile).digest('hex').slice(0, 16)
    : '');
  const greetingSlot = isPublic ? '' : greetingBucket(ctx?.greeting);
  // Canonicalise as JSON of the fields the prompt actually references,
  // in the prompt's ranked order. Stable stringification via an array
  // of tuples keeps field ordering deterministic without relying on
  // JS object-key iteration order. Slice MUST match buildDigestPrompt's
  // slice or the cache key drifts from the prompt content.
  const material = JSON.stringify([
    sensitivity ?? '',
    profileSha,
    greetingSlot,
    isPublic ? 'public' : 'private',
    ...stories.slice(0, MAX_STORIES_PER_USER).map((s) => [
      // hash drives ranking (model emits rankedStoryHashes); without
      // it the cache ignores re-ranking and stale ordering is served.
      typeof s.hash === 'string' ? s.hash.slice(0, 8) : '',
      s.headline ?? '',
      s.threatLevel ?? '',
      s.category ?? '',
      s.country ?? '',
      s.source ?? '',
    ]),
  ]);
  const h = createHash('sha256').update(material).digest('hex').slice(0, 16);
  // userId-slot substitution for public mode β€” one cache row per
  // (sensitivity, story-pool) shared across ALL public readers.
  const userSlot = isPublic ? 'public' : userId;
  return `${userSlot}:${sensitivity}:${h}`;
}

/**
 * Resolve the digest prose object via cache β†’ LLM.
 *
 * Backward-compatible signature: existing 4-arg callers behave like
 * today (no profile/greeting β†’ non-personalised lead). New callers
 * pass `ctx` to enable canonical synthesis with greeting + profile.
 *
 * @param {string} userId
 * @param {Array} stories
 * @param {string} sensitivity
 * @param {{ callLLM: Function; cacheGet: Function; cacheSet: Function }} deps
 * @param {DigestPromptCtx} [ctx]
 */
export async function generateDigestProse(userId, stories, sensitivity, deps, ctx = {}) {
  // v6 key (2026-05-14): bumped from v5 alongside the F6 date-grounding
  // line appended to DIGEST_PROSE_SYSTEM_BASE by buildDigestPrompt.
  // Every v5 row was produced from a prompt with no notion of "today"
  // and may state a fabricated year in the lead/threads/signals β€” the
  // exact bug F6 fixes. validateDigestProseShape revalidates cache
  // hits, but its grounding gate is proper-noun based and does NOT
  // catch date/numeric fabrication, so a v5 row would re-pass and
  // ship for the 4h TTL. Evicting v5 forces regeneration through the
  // date-grounded prompt.
  //
  // v5 (2026-05-12): bumped from v4 alongside the grounding gate in
  // validateDigestProseShape. v4 rows may have been written for
  // shape-valid but content-fabricated leads (May 12 incident: a
  // Trump-era geopolitics pool shipped a "President Biden crypto
  // executive order" fabricated lead that passed the shape-only
  // validator). Evicting v4 forced regeneration through the new
  // grounded gate; ungrounded re-rolls fall through to L2/L3.
  //
  // v4 (2026-04-25 evening): bumped from v3 when the prompt gained
  // a BANNED-phrasing list + "name the specific actor and event"
  // lead instructions, after a regression where evening briefs
  // shipped vapid editorial filler ("the global stage is buzzing",
  // "navigating the evolving landscape"). v3 cache rows still in
  // TTL would otherwise serve stale vapid leads for 4h post-deploy.
  //
  // v7 (2026-05-17): bumped from v6 alongside PR #3751's category
  // persistence. `hashDigestInput` folds `s.category` into the hash
  // material; pre-PR every story carried 'General' (no category was
  // persisted on story:track:v1), post-PR carries the per-story
  // Title-Cased EventCategory value. v6 cache rows would otherwise
  // serve digest prose generated against the pre-PR all-General pool
  // for the full 4h TTL. Sibling bumps applied to whymatters (v4β†’v5)
  // and description (v2β†’v3) β€” all three caches depend on the same
  // story.category field via hashBriefStory / hashDigestInput.
  //
  // v8 (2026-05-18): bumped from v7 when DIGEST_PROSE_SYSTEM_BASE gained
  // anti-stitching instructions (May 17 brief shipped a lead that stapled
  // Ebola + Israel-Lebanon with "This declaration comes as…" β€” two
  // unrelated top stories awkwardly joined). The prompt now explicitly
  // forbids weak temporal connectives ("This comes as", "Meanwhile",
  // "At the same time", "In other news", "Elsewhere", "Across the world",
  // "On another front", "In a separate development") and instructs the
  // model to lead with ONE primary story when two top stories aren't
  // substantively linked. v7 cache rows would otherwise serve stitched
  // leads for the full 4h TTL. Prompt content change β†’ cache invalidation.
  const key = `brief:llm:digest:v8:${hashDigestInput(userId, stories, sensitivity, ctx)}`;
  try {
    const hit = await deps.cacheGet(key);
    // CRITICAL: re-run the shape+grounding validator on cache hits.
    // Without this, a bad row (written under an older buggy code
    // path, partial write, tampered Redis, or shape-valid-but-
    // ungrounded content from a pre-v5 worker that hasn't deployed
    // yet) flows straight into envelope.data.digest and the user
    // sees a hallucinated lead. Treat a validation-failed hit the
    // same as a miss β€” re-LLM and overwrite.
    if (hit) {
      const validated = validateDigestProseShape(hit, stories);
      if (validated) return validated;
    }
  } catch { /* cache miss fine */ }
  const { system, user } = buildDigestPrompt(stories, sensitivity, ctx);
  let text = null;
  try {
    text = await deps.callLLM(system, user, {
      maxTokens: 900,
      temperature: 0.4,
      timeoutMs: 15_000,
      skipProviders: BRIEF_LLM_SKIP_PROVIDERS,
      stage: 'brief-digest-cron',
    });
  } catch (err) {
    // LLM-side failure (timeout, provider down, network). Distinct
    // from "LLM responded but output was malformed/ungrounded" β€”
    // see below.
    console.warn(
      `[brief-llm] digest synthesis: LLM call threw user=${userId} sensitivity=${sensitivity} pool=${stories?.length ?? 0}: ${err?.message ?? 'unknown'}`,
    );
    return null;
  }
  const parsed = parseDigestProse(text, stories);
  if (!parsed) {
    // LLM returned text but parseDigestProse rejected it. Three sub-
    // failures land here, distinguishable on log search:
    //   - text === null/undefined: provider returned no content
    //   - text non-empty but not valid JSON / shape-invalid: model
    //     drift (stripped JSON braces, exceeded length caps)
    //   - shape valid but grounding failed: hallucination rejected
    // On-call triage runs `grep "[brief-llm] digest synthesis"` and
    // distinguishes "LLM threw" (above) vs "ungrounded/malformed
    // output" (here). PR #3667 review round 4 #3 β€” without this log,
    // a sustained model regression is invisible against an infra
    // blip baseline. Cost note: we deliberately do NOT cache the
    // failure (no sentinel write under the v5 key). At temperature
    // 0.4 the next tick may roll a grounded output for the same
    // prompt; caching the failure would block legitimate retries.
    // Cron-level fallback (L1β†’L2β†’L3 in runSynthesisWithFallback)
    // handles the user-visible degradation; this log handles ops
    // visibility.
    const textLen = typeof text === 'string' ? text.length : 0;
    console.warn(
      `[brief-llm] digest synthesis: ungrounded or malformed output user=${userId} sensitivity=${sensitivity} pool=${stories?.length ?? 0} text_len=${textLen}`,
    );
    return null;
  }
  try {
    await deps.cacheSet(key, parsed, DIGEST_PROSE_TTL_SEC);
  } catch { /* ignore */ }
  return parsed;
}

/**
 * Non-personalised wrapper for share-URL surfaces. Strips profile
 * and greeting; substitutes 'public' for userId in the cache key
 * (see hashDigestInput) so all public-share readers of the same
 * (sensitivity, story-pool) hit one cache row.
 *
 * Note the missing `userId` parameter β€” by design. Callers MUST
 * NOT thread their authenticated user's id through this function;
 * the public lead must never carry per-user salt.
 *
 * @param {Array} stories
 * @param {string} sensitivity
 * @param {{ callLLM: Function; cacheGet: Function; cacheSet: Function }} deps
 * @returns {ReturnType<typeof generateDigestProse>}
 */
export async function generateDigestProsePublic(stories, sensitivity, deps) {
  // userId param to generateDigestProse is unused when isPublic=true
  // (see hashDigestInput's userSlot logic). Pass an empty string so
  // a typo on a future caller can't accidentally salt the public
  // cache.
  return generateDigestProse('', stories, sensitivity, deps, {
    profile: null,
    greeting: null,
    isPublic: true,
  });
}

// ── Envelope enrichment ────────────────────────────────────────────────────

/**
 * Bounded-concurrency map. Preserves input order. Doesn't short-circuit
 * on individual failures β€” fn is expected to return a sentinel (null)
 * on error and the caller decides.
 */
async function mapLimit(items, limit, fn) {
  if (!Array.isArray(items) || items.length === 0) return [];
  const n = Math.min(Math.max(1, limit), items.length);
  const out = new Array(items.length);
  let next = 0;
  async function worker() {
    while (true) {
      const idx = next++;
      if (idx >= items.length) return;
      try {
        out[idx] = await fn(items[idx], idx);
      } catch {
        out[idx] = items[idx];
      }
    }
  }
  await Promise.all(Array.from({ length: n }, worker));
  return out;
}

/**
 * Take a baseline BriefEnvelope (stubbed whyMatters + stubbed lead /
 * threads / signals) and enrich it with LLM output. All failures fall
 * through cleanly β€” the envelope that comes out is always a valid
 * BriefEnvelope (structure unchanged; only string/array field
 * contents are substituted).
 *
 * `opts.skipDigestProse` β€” when true, the per-user digest-prose call
 * is SKIPPED entirely and `envelope.data.digest` is passed through
 * untouched; only per-story `whyMatters` / `description` are
 * enriched. The compose path passes this because it has ALREADY
 * produced the canonical synthesis (via `runSynthesisWithFallback`)
 * and spliced it into the envelope. Without the skip, this function
 * re-synthesises here β€” a SECOND, ctx-free `generateDigestProse`
 * call that overwrites the compose-pass synthesis and breaks the
 * compose↔send parity contract. See plan
 * docs/plans/2026-05-14-001-fix-brief-pipeline-parity-grounding-opinion-plan.md
 * (F1, "call site 2") + Codex review R2.
 *
 * @param {object} envelope
 * @param {{ userId: string; sensitivity?: string }} rule
 * @param {{ callLLM: Function; cacheGet: Function; cacheSet: Function }} deps
 * @param {{ skipDigestProse?: boolean }} [opts]
 */
export async function enrichBriefEnvelopeWithLLM(envelope, rule, deps, opts = {}) {
  if (!envelope?.data || !Array.isArray(envelope.data.stories)) return envelope;
  const stories = envelope.data.stories;
  // Default to 'high' (NOT 'all') so the digest prompt and cache key
  // align with what the rest of the pipeline (compose, buildDigest,
  // cache, log) treats undefined-sensitivity rules as. Mismatched
  // defaults would (a) mislead personalization β€” the prompt would say
  // "Reader sensitivity level: all" while the actual brief contains
  // only critical/high stories β€” and (b) bust the cache for legacy
  // rules vs explicit-'all' rules that should share entries. See PR
  // #3387 review (P3).
  const sensitivity = rule?.sensitivity ?? 'high';

  // Per-story enrichment β€” whyMatters AND description in parallel
  // per story (two LLM calls) but bounded across stories.
  const enrichedStories = await mapLimit(stories, WHY_MATTERS_CONCURRENCY, async (story) => {
    const [why, desc] = await Promise.all([
      generateWhyMatters(story, deps),
      generateStoryDescription(story, deps),
    ]);
    if (!why && !desc) return story;
    return {
      ...story,
      ...(why ? { whyMatters: why } : {}),
      ...(desc ? { description: desc } : {}),
    };
  });

  // Per-user digest prose β€” one call, UNLESS the caller already
  // supplied the canonical synthesis (skipDigestProse). See the
  // function-header note: re-synthesising here is the "call site 2"
  // parity regression.
  let digest = envelope.data.digest;
  if (opts?.skipDigestProse !== true) {
    const prose = await generateDigestProse(rule.userId, stories, sensitivity, deps);
    if (prose) {
      digest = {
        ...envelope.data.digest,
        lead: prose.lead,
        threads: prose.threads,
        signals: prose.signals,
      };
    }
  }

  return {
    ...envelope,
    data: {
      ...envelope.data,
      digest,
      stories: enrichedStories,
    },
  };
}