fix: a greeting is not a query, so nothing is searched for one
"hello" came back with four confident citations and a list of topics the learner might like, drawn from questions about caustic ingestion. Retrieval had done nothing wrong: reciprocal-rank fusion always returns an order, and the similarity gate that exists to catch this is a threshold — "hi" embeds at 0.46 against a corpus of clinical prose and is caught, "ok" at 0.51 and "good morning" at 0.50 are not. So the decision is made before any measuring: a message made entirely of pleasantries, or asking what the assistant is, gets mode "chat" — no retrieval at all, no shortlist to cite from, and a prompt that says what it can do without claiming to know what is in the library, because it has not looked. The vocabulary is closed rather than a length rule, so "croup dose?" is still a query. A missed greeting costs a slightly odd reply; a swallowed question costs an answer. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01TqXevQJhxFrM7jJg82cgZN
This commit is contained in:
parent
177df84817
commit
ce8074cbe3
3 changed files with 124 additions and 10 deletions
|
|
@ -168,14 +168,21 @@ async def ask(conversation_id: int, data: AskIn, db: Session = Depends(get_db),
|
||||||
if not model_id:
|
if not model_id:
|
||||||
raise HTTPException(503, "No AI model is configured. Ask an admin to set one up.")
|
raise HTTPException(503, "No AI model is configured. Ask an admin to set one up.")
|
||||||
|
|
||||||
sources = ai_mode_service.retrieve(db, current_user, question)
|
# A greeting is not a query, and searching a clinical corpus for one comes
|
||||||
# How close the nearest thing in the library actually is, which decides
|
# back full of confident nonsense. Decided before retrieval so nothing is
|
||||||
# which of the three answers this question gets. The shortlist alone cannot
|
# searched for, rather than searched for and then thrown away.
|
||||||
# tell you: reciprocal-rank fusion throws the distances away and returns an
|
if ai_mode_service.is_small_talk(question):
|
||||||
# order that is never empty, so a question about photosynthesis came back
|
sources, similarity, mode = [], None, "chat"
|
||||||
# with six paediatric sources and an instruction to answer only from them.
|
else:
|
||||||
similarity = ai_mode_service.closeness(db, question)
|
sources = ai_mode_service.retrieve(db, current_user, question)
|
||||||
mode = ai_mode_service.answer_mode(similarity, sources)
|
# How close the nearest thing in the library actually is, which decides
|
||||||
|
# which of the three answers this question gets. The shortlist alone
|
||||||
|
# cannot tell you: reciprocal-rank fusion throws the distances away and
|
||||||
|
# returns an order that is never empty, so a question about
|
||||||
|
# photosynthesis came back with six paediatric sources and an
|
||||||
|
# instruction to answer only from them.
|
||||||
|
similarity = ai_mode_service.closeness(db, question)
|
||||||
|
mode = ai_mode_service.answer_mode(similarity, sources, question)
|
||||||
if mode == "open":
|
if mode == "open":
|
||||||
# Nothing to cite, so nothing is offered for citation — the shortlist is
|
# Nothing to cite, so nothing is offered for citation — the shortlist is
|
||||||
# not passed to a model that has just been told the library does not
|
# not passed to a model that has just been told the library does not
|
||||||
|
|
|
||||||
|
|
@ -300,8 +300,72 @@ def closeness(db: Session, query: str) -> float | None:
|
||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def answer_mode(similarity: float | None, sources: list[dict]) -> str:
|
#: Words a message can be made entirely of and still not be a question about
|
||||||
"""Which of the three answers this question gets: sourced, adjacent, or open.
|
#: anything. Kept as a closed list rather than a length rule: "croup dose?" is
|
||||||
|
#: two words and is very much a query, while "ok thanks" is two words and is
|
||||||
|
#: not.
|
||||||
|
SMALL_TALK_WORDS = {
|
||||||
|
"hi", "hello", "hey", "yo", "hiya", "greetings", "howdy", "morning",
|
||||||
|
"afternoon", "evening", "good", "day", "night", "sup", "hallo",
|
||||||
|
"thanks", "thank", "thx", "ty", "cheers", "appreciated", "much", "you",
|
||||||
|
"ok", "okay", "okey", "k", "sure", "cool", "nice", "great", "perfect",
|
||||||
|
"yes", "yeah", "yep", "no", "nope", "nah", "please", "sorry", "welcome",
|
||||||
|
"bye", "goodbye", "later", "see", "ya", "ciao",
|
||||||
|
"lol", "haha", "hmm", "hm", "oh", "ah", "wow", "test", "testing",
|
||||||
|
}
|
||||||
|
|
||||||
|
#: Questions about the assistant itself. The library has nothing to say about
|
||||||
|
#: these by construction, and searching it for them produces exactly the
|
||||||
|
#: nonsense this guard exists to stop: a greeting answered with four citations.
|
||||||
|
ABOUT_ASSISTANT_RE = re.compile(
|
||||||
|
r"^\s*(who\s+(are|r)\s+(you|u)|what\s+(are|r)\s+(you|u)|"
|
||||||
|
r"what\s+(can|do)\s+(you|u)\s+(do|help)|how\s+(do|does)\s+(this|it)\s+work|"
|
||||||
|
r"what\s+is\s+this|help)\b",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def is_small_talk(text: str) -> bool:
|
||||||
|
"""Whether this message is not a query at all.
|
||||||
|
|
||||||
|
A greeting is not a low-scoring question, and the difference matters:
|
||||||
|
reciprocal-rank fusion always returns an order, so "hello" comes back with
|
||||||
|
six paediatric sources ranked confidently against nothing. The similarity
|
||||||
|
gate below catches most of that, but it is a threshold, and "ok" and "good
|
||||||
|
morning" happen to land the wrong side of it — a greeting embedded into the
|
||||||
|
same space as a corpus of clinical prose scores wherever it scores. So this
|
||||||
|
is decided before any measuring is done, on the text itself.
|
||||||
|
|
||||||
|
Deliberately narrow. Anything with a word in it that is not pleasantry
|
||||||
|
falls through to retrieval, because the cost of a missed greeting is a
|
||||||
|
slightly odd reply and the cost of a swallowed question is an unanswered
|
||||||
|
one.
|
||||||
|
"""
|
||||||
|
stripped = (text or "").strip()
|
||||||
|
if not stripped:
|
||||||
|
return True
|
||||||
|
if ABOUT_ASSISTANT_RE.match(stripped):
|
||||||
|
return True
|
||||||
|
words = re.findall(r"[a-z]+", stripped.lower())
|
||||||
|
if not words or len(words) > 5:
|
||||||
|
return False
|
||||||
|
return all(word in SMALL_TALK_WORDS for word in words)
|
||||||
|
|
||||||
|
|
||||||
|
CHAT_PROMPT = (
|
||||||
|
ROLE +
|
||||||
|
"The learner has not asked a question yet — this turn is a greeting, a "
|
||||||
|
"thank-you, or a question about you rather than about medicine.\n\n"
|
||||||
|
"Reply in one or two short sentences. Say what you can do: answer from the "
|
||||||
|
"articles, questions and cards in their library, and point them at "
|
||||||
|
"questions to practise. Do not list topics, do not cite anything, and do "
|
||||||
|
"not invent what their library contains — you have not looked."
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def answer_mode(similarity: float | None, sources: list[dict],
|
||||||
|
question: str | None = None) -> str:
|
||||||
|
"""Which answer this turn gets: chat, sourced, adjacent, or open.
|
||||||
|
|
||||||
Decided by a number rather than by asking the model to work out which
|
Decided by a number rather than by asking the model to work out which
|
||||||
situation it is in. Classification written as prose in a prompt is the part
|
situation it is in. Classification written as prose in a prompt is the part
|
||||||
|
|
@ -312,6 +376,8 @@ def answer_mode(similarity: float | None, sources: list[dict]) -> str:
|
||||||
because the ruler is missing would silently drop every citation on a
|
because the ruler is missing would silently drop every citation on a
|
||||||
deployment where semantic search happens to be unavailable.
|
deployment where semantic search happens to be unavailable.
|
||||||
"""
|
"""
|
||||||
|
if question is not None and is_small_talk(question):
|
||||||
|
return "chat"
|
||||||
if not sources:
|
if not sources:
|
||||||
return "open"
|
return "open"
|
||||||
if similarity is None or similarity >= STRONG_MATCH:
|
if similarity is None or similarity >= STRONG_MATCH:
|
||||||
|
|
@ -320,6 +386,11 @@ def answer_mode(similarity: float | None, sources: list[dict]) -> str:
|
||||||
|
|
||||||
|
|
||||||
def build_prompt(sources: list[dict], mode: str = "sourced") -> str:
|
def build_prompt(sources: list[dict], mode: str = "sourced") -> str:
|
||||||
|
if mode == "chat":
|
||||||
|
# No sources, and — unlike "open" — no announcement that the library
|
||||||
|
# does not cover it either. Nobody who says hello is waiting to be told
|
||||||
|
# what their library lacks.
|
||||||
|
return CHAT_PROMPT
|
||||||
if mode == "open" or not sources:
|
if mode == "open" or not sources:
|
||||||
# Nothing in the library is close, so the honest answer is to say that
|
# Nothing in the library is close, so the honest answer is to say that
|
||||||
# and then help anyway. Refusing outright was the old behaviour and it
|
# and then help anyway. Refusing outright was the old behaviour and it
|
||||||
|
|
|
||||||
|
|
@ -90,6 +90,42 @@ class CitationContractTests(unittest.TestCase):
|
||||||
# dropping every citation because the ruler is missing would be worse.
|
# dropping every citation because the ruler is missing would be worse.
|
||||||
self.assertEqual(ai_mode_service.answer_mode(None, sources), "sourced")
|
self.assertEqual(ai_mode_service.answer_mode(None, sources), "sourced")
|
||||||
|
|
||||||
|
def test_a_greeting_is_not_a_query(self):
|
||||||
|
"""Whatever it scores, "hello" is not a question about anything.
|
||||||
|
|
||||||
|
This is the bug in the screenshot: a greeting came back with four
|
||||||
|
confident citations, because reciprocal-rank fusion always returns an
|
||||||
|
order and "ok" happens to embed at 0.51 against a corpus of clinical
|
||||||
|
prose — the wrong side of the adjacency threshold.
|
||||||
|
"""
|
||||||
|
sources = self.sources()
|
||||||
|
for greeting in ["hi", "Hello", "hello!", " thanks ", "ok thanks",
|
||||||
|
"good morning", "Thank you!", "bye", "",
|
||||||
|
"who are you?", "what can you do"]:
|
||||||
|
self.assertTrue(ai_mode_service.is_small_talk(greeting), greeting)
|
||||||
|
self.assertEqual(
|
||||||
|
ai_mode_service.answer_mode(0.9, sources, greeting), "chat", greeting)
|
||||||
|
|
||||||
|
def test_a_short_clinical_question_is_still_a_query(self):
|
||||||
|
# The guard is a closed vocabulary rather than a length rule, precisely
|
||||||
|
# so that these keep reaching retrieval.
|
||||||
|
sources = self.sources()
|
||||||
|
for query in ["croup dose?", "ok to give ibuprofen at 3 months?",
|
||||||
|
"no stridor now what", "hi flow nasal cannula",
|
||||||
|
"thanks to which vaccine has Hib fallen?"]:
|
||||||
|
self.assertFalse(ai_mode_service.is_small_talk(query), query)
|
||||||
|
self.assertEqual(
|
||||||
|
ai_mode_service.answer_mode(0.72, sources, query), "sourced", query)
|
||||||
|
|
||||||
|
def test_the_chat_prompt_offers_nothing_and_claims_nothing(self):
|
||||||
|
prompt = ai_mode_service.build_prompt(self.sources(), "chat")
|
||||||
|
# No shortlist reaches a turn that was never a query, so there is
|
||||||
|
# nothing for the model to cite even if it tries.
|
||||||
|
self.assertNotIn("[[section:7#abc]]", prompt)
|
||||||
|
self.assertNotIn("SOURCES", prompt)
|
||||||
|
# And unlike "open", it does not announce a gap nobody asked about.
|
||||||
|
self.assertNotIn("Nothing in this learner's library", prompt)
|
||||||
|
|
||||||
|
|
||||||
class _AiModeBase(unittest.TestCase):
|
class _AiModeBase(unittest.TestCase):
|
||||||
"""Fixtures shared by the route and retrieval cases; holds no tests itself."""
|
"""Fixtures shared by the route and retrieval cases; holds no tests itself."""
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue