cazyundee commited on
Commit
be92d5f
·
1 Parent(s): d409fec

sync: from ael backend/space/

Browse files
Files changed (1) hide show
  1. rerank.py +50 -5
rerank.py CHANGED
@@ -78,9 +78,23 @@ DROP_BELOW = float(os.environ.get("AEL_RERANK_DROP", "0.5"))
78
  # one, not a target: a short page is fine, an empty one is not.
79
  MIN_RESULTS = 1
80
 
 
 
 
 
 
 
 
 
 
 
81
  # Recent (query, scores) for threshold tuning. Bounded so a long-lived Space
82
  # cannot grow without limit.
83
  RECENT = deque(maxlen=int(os.environ.get("AEL_RERANK_LOG", "200")))
 
 
 
 
84
 
85
  _router = None
86
  _router_lock = threading.Lock()
@@ -197,7 +211,10 @@ def status():
197
  def recent_scores(limit=20):
198
  """Recent (query, scores) pairs, newest first — the threshold-tuning data."""
199
  items = list(RECENT)[-limit:][::-1]
200
- return [{"q": q, "scores": s} for q, s in items]
 
 
 
201
 
202
 
203
  def _question(query):
@@ -270,7 +287,24 @@ def rerank(query, results, budget_ms=800):
270
 
271
  t0 = time.time()
272
  scored = []
 
273
  for i, r in enumerate(results):
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
274
  passage = _passage(r)
275
  if not passage:
276
  # No text to judge. Keep it, unscored, and let the ordering stand.
@@ -282,11 +316,18 @@ def rerank(query, results, budget_ms=800):
282
  except Exception as e: # noqa: BLE001
283
  print(f"[rerank] score failed for result {i}: {type(e).__name__}: {e}", flush=True)
284
  return _stamp(dict(out, results=results)) # fail open, whole set untouched
285
- if (time.time() - t0) * 1000 > budget_ms:
286
- # Over budget: return what we have not yet reordered, in order.
287
- print(f"[rerank] over budget ({budget_ms}ms) after {i} results", flush=True)
288
- return _stamp(dict(out, results=results))
289
  scored.append((p, i, r))
 
 
 
 
 
 
 
 
 
 
 
290
 
291
  if not any(p is not None for p, _, _ in scored):
292
  return _stamp(out)
@@ -303,6 +344,10 @@ def rerank(query, results, budget_ms=800):
303
  if p is not None
304
  ]
305
  RECENT.append((query[:120], [p for p, _, _ in scored if p is not None]))
 
 
 
 
306
 
307
  if dropped and len(kept) < MIN_RESULTS:
308
  # The model rejected everything, which is a model that did not
 
78
  # one, not a target: a short page is fine, an empty one is not.
79
  MIN_RESULTS = 1
80
 
81
+ # How many results the model is allowed to judge. A 421M cross-encoder on a
82
+ # shared CPU does not score eight documents inside an 800ms budget, and the
83
+ # first version's answer to that was to throw the whole set away — which is why
84
+ # a reranker that was installed, loaded, healthy and simply too slow reported
85
+ # itself identically to one that had never run. Judging the first N and leaving
86
+ # the rest in engine order is strictly better than judging none: the survivors
87
+ # are reordered by relevance and the rejects are dropped, and the untested tail
88
+ # keeps the order the race gave it.
89
+ MAX_SCORED = int(os.environ.get("AEL_RERANK_MAX", "6"))
90
+
91
  # Recent (query, scores) for threshold tuning. Bounded so a long-lived Space
92
  # cannot grow without limit.
93
  RECENT = deque(maxlen=int(os.environ.get("AEL_RERANK_LOG", "200")))
94
+ # Wall-clock of recent reranks, same bound. This is the tuning data for
95
+ # MAX_SCORED and the budget: the scores say what to keep, this says what it
96
+ # costs to judge it.
97
+ RECENT_MS = deque(maxlen=int(os.environ.get("AEL_RERANK_LOG", "200")))
98
 
99
  _router = None
100
  _router_lock = threading.Lock()
 
211
  def recent_scores(limit=20):
212
  """Recent (query, scores) pairs, newest first — the threshold-tuning data."""
213
  items = list(RECENT)[-limit:][::-1]
214
+ ms = list(RECENT_MS)[-limit:][::-1]
215
+ return [
216
+ {"q": q, "scores": s, "ms": m} for (q, s), m in zip(items, ms)
217
+ ]
218
 
219
 
220
  def _question(query):
 
287
 
288
  t0 = time.time()
289
  scored = []
290
+ judged = 0
291
  for i, r in enumerate(results):
292
+ if judged >= MAX_SCORED:
293
+ # Enough judged to reorder by. The tail is kept, unscored, at the
294
+ # end — we have no opinion about it and will not pretend to.
295
+ scored.append((None, i, r))
296
+ continue
297
+ if judged and (time.time() - t0) * 1000 > budget_ms:
298
+ # Over budget, but not empty-handed. A partial judgement is worth
299
+ # more than none: the documents we did score get reordered and the
300
+ # ones the model rejected get dropped. Only a run that scored
301
+ # *nothing* throws its work away.
302
+ print(
303
+ f"[rerank] over budget ({budget_ms}ms) after {judged} of {len(results)}"
304
+ f" — reranking what was judged",
305
+ flush=True,
306
+ )
307
+ break
308
  passage = _passage(r)
309
  if not passage:
310
  # No text to judge. Keep it, unscored, and let the ordering stand.
 
316
  except Exception as e: # noqa: BLE001
317
  print(f"[rerank] score failed for result {i}: {type(e).__name__}: {e}", flush=True)
318
  return _stamp(dict(out, results=results)) # fail open, whole set untouched
 
 
 
 
319
  scored.append((p, i, r))
320
+ judged += 1
321
+
322
+ # Anything the loop never reached keeps its place at the end, unscored.
323
+ if len(scored) < len(results):
324
+ seen_idx = {i for _, i, _ in scored}
325
+ for i, r in enumerate(results):
326
+ if i not in seen_idx:
327
+ scored.append((None, i, r))
328
+
329
+ out["ms"] = int((time.time() - t0) * 1000)
330
+ out["judged"] = judged
331
 
332
  if not any(p is not None for p, _, _ in scored):
333
  return _stamp(out)
 
344
  if p is not None
345
  ]
346
  RECENT.append((query[:120], [p for p, _, _ in scored if p is not None]))
347
+ # The duration is the number this stage actually needs tuning against, and
348
+ # it is the one that was missing: a reranker that is too slow for its own
349
+ # budget and a reranker that is not installed look identical from outside.
350
+ RECENT_MS.append(out["ms"])
351
 
352
  if dropped and len(kept) < MIN_RESULTS:
353
  # The model rejected everything, which is a model that did not