| """Dependency parsing — spaCy-driven slots (no synonym or domain word lists)."""
|
|
|
| from __future__ import annotations
|
|
|
| import re
|
|
|
| from app.engine.classify import classify_sentence
|
| from app.engine.models import SentenceSlots
|
| from app.pipeline.nlp import get_nlp
|
|
|
|
|
| def _span_text(tokens: list) -> str:
|
| if not tokens:
|
| return ""
|
| return "".join(t.text_with_ws for t in tokens).strip()
|
|
|
|
|
| def _subtree_tokens(token) -> list:
|
| return sorted(token.subtree, key=lambda t: t.i)
|
|
|
|
|
| def _under_duration_prep(token) -> bool:
|
| """True if token sits under a duration-like preposition (for/over/within/during)."""
|
| cur = token
|
| while cur is not None and cur.head != cur:
|
| if cur.dep_ == "prep" and cur.lemma_.lower() in {"for", "over", "within", "during"}:
|
|
|
| return True
|
| cur = cur.head
|
| return False
|
|
|
|
|
| def _is_duration_ent(ent, doc_text: str) -> bool:
|
| """DATE/TIME under for/over, or minute/hour-style spans — not sentence adjuncts."""
|
| text = ent.text.strip()
|
| if re.search(
|
| rf"\b(for|over|within|during)\s+{re.escape(text)}\b",
|
| doc_text,
|
| flags=re.I,
|
| ):
|
| return True
|
| for t in ent:
|
| if _under_duration_prep(t):
|
|
|
| if ent.label_ in {"TIME", "DATE"} and re.search(
|
| r"\b(minute|minutes|hour|hours|second|seconds)\b",
|
| text,
|
| flags=re.I,
|
| ):
|
| return True
|
| if re.match(r"at\s+least\b", text, flags=re.I):
|
| return True
|
| if ent.label_ == "TIME" and _under_duration_prep(t):
|
| return True
|
| return False
|
|
|
|
|
| def _is_frontable_adv(tok, root) -> bool:
|
| """ADV modifiers that can move — derived from POS/DEP, not word lists."""
|
| if tok.pos_ != "ADV":
|
| return False
|
| if tok.dep_ not in {"advmod", "npadvmod"}:
|
| return False
|
| if tok.lemma_.lower() in {"not", "n't"} or tok.dep_ == "neg":
|
| return False
|
|
|
| if tok.head.pos_ in {"ADJ", "ADV"} and tok.head.i != root.i:
|
| return False
|
| return True
|
|
|
|
|
| def _is_temporal_tok(tok) -> bool:
|
| if tok.ent_type_ in {"DATE", "TIME"}:
|
| return True
|
|
|
| if tok.dep_ in {"npadvmod", "advmod"} and tok.ent_type_ in {"DATE", "TIME"}:
|
| return True
|
| if tok.dep_ == "npadvmod" and tok.pos_ in {"NOUN", "PROPN"}:
|
|
|
| if tok.head.pos_ in {"VERB", "AUX"}:
|
| return True
|
| return False
|
|
|
|
|
| def _dedupe_spans(spans: list[str]) -> list[str]:
|
| cleaned = [s.strip() for s in spans if s and s.strip()]
|
| cleaned.sort(key=len, reverse=True)
|
| kept: list[str] = []
|
| for s in cleaned:
|
| low = s.lower()
|
| if any(low != k.lower() and low in k.lower() for k in kept):
|
| continue
|
| if any(k.lower() == low for k in kept):
|
| continue
|
| kept.append(s)
|
| return kept
|
|
|
|
|
| def _freq_from_doc(doc) -> str:
|
| """every/each + noun via dependency structure (not a word list of weekdays)."""
|
| for tok in doc:
|
| if tok.pos_ == "DET" and tok.lemma_.lower() in {"every", "each"}:
|
| head = tok.head
|
| if head.pos_ in {"NOUN", "PROPN"}:
|
| return _span_text([tok] + _subtree_tokens(head))
|
|
|
| if tok.pos_ == "ADV" and tok.dep_ in {"advmod", "npadvmod"}:
|
|
|
| if "day" in tok.lemma_.lower() or tok.text.lower().endswith("day"):
|
| if tok.ent_type_ in {"DATE", "TIME", ""} and len(tok.text) > 3:
|
|
|
| if not _under_duration_prep(tok):
|
|
|
| if tok.head.pos_ in {"VERB", "AUX"}:
|
| return tok.text
|
| return ""
|
|
|
|
|
| def extract_slots(text: str) -> SentenceSlots:
|
| """Parse subject/verb/object/time/place/manner/negation/entities via spaCy."""
|
| raw = (text or "").strip()
|
| slots = SentenceSlots(text=raw, sentence_type=classify_sentence(raw))
|
|
|
| nlp = get_nlp()
|
| if nlp is None:
|
| return _extract_slots_regex(raw)
|
|
|
| doc = nlp(raw)
|
| slots.entities = [
|
| ent.text
|
| for ent in doc.ents
|
| if ent.label_ in {
|
| "PERSON", "ORG", "GPE", "LOC", "DATE", "TIME", "MONEY", "PERCENT", "CARDINAL",
|
| }
|
| ]
|
| if slots.sentence_type not in {"simple_declarative", "compound"}:
|
| slots.auxiliaries = [token.text for token in doc if token.pos_ == "AUX"]
|
| slots.reasons.append(f"skip:{slots.sentence_type}")
|
| return slots
|
|
|
| root = next((t for t in doc if t.dep_ == "ROOT" and t.pos_ in {"VERB", "AUX"}), None)
|
| if root is None:
|
| root = next((t for t in doc if t.pos_ == "VERB"), None)
|
| if root is None:
|
| slots.reasons.append("no_verb")
|
| return slots
|
|
|
|
|
| subj_child = None
|
| for child in root.children:
|
| if child.dep_ in {"nsubj", "nsubjpass", "csubj", "csubjpass"}:
|
| subj_child = child
|
| break
|
|
|
|
|
| early_manner: list = []
|
| for t in doc:
|
| if _is_frontable_adv(t, root):
|
| if not (_is_temporal_tok(t) or t.ent_type_ in {"DATE", "TIME"}):
|
| if t.dep_ != "npadvmod" or t.pos_ not in {"NOUN", "PROPN"}:
|
| early_manner.append(t)
|
|
|
| if subj_child is not None:
|
| exclude = {t.i for t in early_manner}
|
| subj_toks = [t for t in _subtree_tokens(subj_child) if t.i not in exclude]
|
| slots.subject = _span_text(subj_toks)
|
| slots.subject_is_proper = subj_child.pos_ == "PROPN" or (
|
| subj_child.tag_ == "NN"
|
| and slots.subject[:1].isupper()
|
| and not any(token.dep_ == "det" for token in subj_toks)
|
| )
|
| else:
|
| slots.subject = ""
|
|
|
| neg_parts = [
|
| t.text
|
| for t in doc
|
| if t.dep_ == "neg" and (t.head == root or t.head.head == root)
|
| ]
|
| slots.negation = " ".join(neg_parts)
|
|
|
| verb_toks = [t for t in root.lefts if t.dep_ in {"aux", "auxpass", "neg"}] + [root]
|
| slots.verb = root.text
|
| slots.verb_phrase = _span_text(sorted(verb_toks, key=lambda t: t.i))
|
| slots.auxiliaries = [t.text for t in verb_toks if t.pos_ == "AUX"]
|
| slots.verb_starts_with_aux = root.pos_ == "AUX" or any(
|
| t.dep_ in {"aux", "auxpass"} for t in verb_toks
|
| )
|
|
|
| obj_toks: list = []
|
| for child in root.children:
|
| if child.dep_ in {"dobj", "obj", "attr"}:
|
| obj_toks = _subtree_tokens(child)
|
| break
|
|
|
| for child in root.children:
|
| if child.dep_ in {"xcomp", "ccomp"}:
|
| extra = _subtree_tokens(child)
|
|
|
| for t in extra:
|
| if t not in obj_toks:
|
| obj_toks.append(t)
|
| obj_toks = sorted(obj_toks, key=lambda t: t.i)
|
| if not obj_toks:
|
| for child in root.children:
|
| if child.dep_ in {"xcomp", "ccomp"}:
|
| obj_toks = _subtree_tokens(child)
|
| break
|
| slots.object = _span_text(obj_toks)
|
|
|
|
|
| place_parts: list[str] = []
|
| for t in doc:
|
| if t.dep_ != "prep" or t.pos_ != "ADP":
|
| continue
|
|
|
| if t.head.pos_ in {"NOUN", "PROPN", "NUM", "PRON"} and t.lemma_.lower() == "of":
|
| continue
|
| if t.head != root and t.head.dep_ not in {"attr", "dobj", "obj"}:
|
| continue
|
| span = _span_text(_subtree_tokens(t))
|
| pobj = next((c for c in t.children if c.dep_ == "pobj"), None)
|
| if (
|
| pobj is not None
|
| and t.lemma_.lower() in {"for", "over", "within", "during"}
|
| and any(c.like_num or c.ent_type_ in {"TIME", "CARDINAL"} for c in pobj.subtree)
|
| ):
|
| continue
|
| if t.head == root or t.head.dep_ in {"attr", "dobj", "obj"}:
|
| place_parts.append(span)
|
| slots.place = _dedupe_spans(place_parts)[0] if place_parts else ""
|
|
|
| time_parts: list[str] = []
|
| manner_parts: list[str] = []
|
|
|
| for ent in doc.ents:
|
| if ent.label_ not in {"DATE", "TIME"}:
|
| continue
|
| if _is_duration_ent(ent, raw):
|
| continue
|
| under = any(_under_duration_prep(t) for t in ent)
|
| if under and re.search(
|
| r"\b(minute|minutes|hour|hours|second|seconds)\b", ent.text, flags=re.I
|
| ):
|
| continue
|
| if under and re.match(r"at\s+least\b", ent.text, flags=re.I):
|
| continue
|
|
|
| if under and ent.label_ == "TIME":
|
| continue
|
| governing_prep = ent.root.head
|
| if governing_prep.dep_ == "prep" and governing_prep.pos_ == "ADP":
|
| time_parts.append(_span_text(_subtree_tokens(governing_prep)))
|
| else:
|
| time_parts.append(ent.text)
|
|
|
| for t in doc:
|
| if _is_frontable_adv(t, root):
|
| if _is_temporal_tok(t) or t.ent_type_ in {"DATE", "TIME"}:
|
| if not _under_duration_prep(t):
|
| time_parts.append(_span_text(_subtree_tokens(t)))
|
| elif t.dep_ == "npadvmod" and t.pos_ in {"NOUN", "PROPN"} and t.head == root:
|
| if not _under_duration_prep(t):
|
| time_parts.append(_span_text(_subtree_tokens(t)))
|
| else:
|
|
|
| manner_parts.append(t.text)
|
|
|
| freq = _freq_from_doc(doc)
|
| if freq and not re.search(r"\b(minute|minutes|hour|hours|second|seconds)\b", freq, flags=re.I):
|
| if not re.search(rf"\b(for|over|within|during)\s+{re.escape(freq)}\b", raw, flags=re.I):
|
| time_parts.append(freq)
|
|
|
|
|
| for t in doc:
|
| if t.dep_ == "npadvmod" and t.head == root and t.pos_ in {"NOUN", "PROPN"}:
|
| if not _under_duration_prep(t):
|
| time_parts.append(_span_text(_subtree_tokens(t)))
|
|
|
| time_parts = _dedupe_spans(list(dict.fromkeys(time_parts)))
|
|
|
| time_parts = [
|
| p
|
| for p in time_parts
|
| if not re.search(r"\b(minute|minutes|hour|hours|second|seconds)\b", p, flags=re.I)
|
| and not re.match(r"at\s+least\b", p, flags=re.I)
|
| and not re.search(rf"\b(for|over|within|during)\s+{re.escape(p)}\b", raw, flags=re.I)
|
| ]
|
| slots.time = time_parts[0] if time_parts else ""
|
|
|
| ly = [m for m in dict.fromkeys(manner_parts) if m.lower().endswith("ly")]
|
| if ly:
|
| slots.manner = ly[0]
|
| else:
|
| slots.manner = next(iter(dict.fromkeys(manner_parts)), "")
|
|
|
| if slots.manner and slots.time and slots.manner.lower() == slots.time.lower():
|
| slots.manner = ""
|
| if slots.manner and slots.time and slots.manner.lower() in slots.time.lower():
|
| slots.manner = ""
|
| residual_adverbs = [
|
| token
|
| for token in early_manner
|
| if token.text.lower() != slots.manner.lower()
|
| and not (_is_temporal_tok(token) or token.ent_type_ in {"DATE", "TIME"})
|
| ]
|
| if residual_adverbs:
|
| phrase_tokens = sorted(
|
| {token.i: token for token in [*verb_toks, *residual_adverbs]}.values(),
|
| key=lambda token: token.i,
|
| )
|
| slots.verb_phrase = _span_text(phrase_tokens)
|
|
|
|
|
| if slots.place and slots.object and slots.place in slots.object:
|
|
|
| pass
|
|
|
| for piece in (slots.time, slots.manner, freq):
|
| if not piece:
|
| continue
|
| for attr in ("object", "subject", "place", "verb_phrase"):
|
| val = getattr(slots, attr) or ""
|
| if piece.lower() in val.lower():
|
| cleaned = re.sub(
|
| rf"\b{re.escape(piece)}\b", "", val, count=1, flags=re.I
|
| ).strip(" ,")
|
| cleaned = re.sub(r"\s+", " ", cleaned).strip()
|
| setattr(slots, attr, cleaned)
|
|
|
| score = 0.2
|
| if slots.subject:
|
| score += 0.25
|
| if slots.verb_phrase:
|
| score += 0.2
|
| if slots.place or slots.object:
|
| score += 0.15
|
| if slots.time:
|
| score += 0.1
|
| if slots.manner:
|
| score += 0.1
|
| slots.confidence = min(1.0, score)
|
| if slots.confidence < 0.45 or not slots.subject or not slots.verb_phrase:
|
| slots.reasons.append("low_confidence")
|
| return slots
|
|
|
|
|
| def _extract_slots_regex(text: str) -> SentenceSlots:
|
| """Minimal regex fallback — structure only, no domain lexicons."""
|
| slots = SentenceSlots(text=text, sentence_type=classify_sentence(text))
|
| if slots.sentence_type != "simple_declarative":
|
| return slots
|
| core = text
|
| m_end = re.search(r"[.!?]+$", text)
|
| if m_end:
|
| core = text[: m_end.start()]
|
|
|
|
|
| manner = ""
|
| mm = re.search(r"\b(\w+ly)$", core, flags=re.I)
|
| if mm:
|
| manner = mm.group(1)
|
| core = core[: mm.start()].strip()
|
|
|
|
|
| time = ""
|
| mf = re.search(r"\b((?:every|each)\s+\w+)$", core, flags=re.I)
|
| if mf:
|
| time = mf.group(1)
|
| core = core[: mf.start()].strip()
|
|
|
| m = re.match(
|
| r"^(?P<subj>.+?)\s+(?P<verb>\w+(?:\s+\w+)?)\s+(?P<rest>.+)$",
|
| core,
|
| flags=re.I,
|
| )
|
| if not m:
|
|
|
| m2 = re.match(r"^(?P<subj>.+?)\s+(?P<verb>\w+)$", core, flags=re.I)
|
| if not m2:
|
| slots.reasons.append("regex_no_match")
|
| return slots
|
| slots.subject = m2.group("subj").strip()
|
| slots.verb_phrase = m2.group("verb").strip()
|
| slots.verb = slots.verb_phrase.split()[-1]
|
| else:
|
| slots.subject = m.group("subj").strip()
|
| slots.verb_phrase = m.group("verb").strip()
|
| slots.verb = slots.verb_phrase.split()[-1]
|
| rest = m.group("rest").strip()
|
| if re.match(r"^to\s+\w+", rest, flags=re.I):
|
| slots.place = rest if " for " not in rest.lower() else ""
|
| slots.object = rest if not slots.place else ""
|
| else:
|
| slots.object = rest
|
|
|
| slots.time = time
|
| slots.manner = manner
|
| score = 0.2 + (0.25 if slots.subject else 0) + (0.2 if slots.verb_phrase else 0)
|
| score += 0.15 if (slots.object or slots.place) else 0
|
| score += 0.1 if slots.time else 0
|
| score += 0.1 if slots.manner else 0
|
| slots.confidence = min(1.0, score)
|
| if slots.confidence < 0.45 or not slots.subject or not slots.verb_phrase:
|
| slots.reasons.append("low_confidence")
|
| return slots
|
|
|