chore: log where each answer's time goes — rewrite, search, first token, total
Numbers only, one line per streamed question, so "is the chat slow?" can be answered from the log instead of from a feeling. Nothing about the request is logged beyond the counts. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Dv6sqaY6Vq3ChZHMem3cnU
This commit is contained in:
parent
d94dc0e357
commit
9c2b067745
1 changed files with 15 additions and 1 deletions
|
|
@ -423,6 +423,7 @@ router.post('/clinical-assistant/chat/stream', async function(req, res) {
|
||||||
|
|
||||||
var toolset = assistantToolset(prepared);
|
var toolset = assistantToolset(prepared);
|
||||||
if (prepared.delegateVision) sendEvent('status', { message: 'Looking at your image…' });
|
if (prepared.delegateVision) sendEvent('status', { message: 'Looking at your image…' });
|
||||||
|
var firstTokenAt = 0;
|
||||||
var ai = await callAIStream(prepared.messages, assistantGenerationOptions({
|
var ai = await callAIStream(prepared.messages, assistantGenerationOptions({
|
||||||
model: prepared.chatModel || undefined,
|
model: prepared.chatModel || undefined,
|
||||||
temperature: 0.15,
|
temperature: 0.15,
|
||||||
|
|
@ -430,8 +431,13 @@ router.post('/clinical-assistant/chat/stream', async function(req, res) {
|
||||||
maxTokens: 2600,
|
maxTokens: 2600,
|
||||||
images: toolset.images
|
images: toolset.images
|
||||||
}), function(delta) {
|
}), function(delta) {
|
||||||
|
if (!firstTokenAt) firstTokenAt = Date.now();
|
||||||
sendEvent('token', { token: delta });
|
sendEvent('token', { token: delta });
|
||||||
});
|
});
|
||||||
|
var t = prepared.timing || {};
|
||||||
|
console.info('[clinical-assistant] timing ms: rewrite=' + (t.rewriteMs || 0) + ' search=' + (t.searchMs || 0) +
|
||||||
|
' first_token=' + (firstTokenAt ? firstTokenAt - started : 0) + ' total=' + (Date.now() - started) +
|
||||||
|
' sources=' + (prepared.sources || []).length + ' model=' + (ai.model || prepared.chatModel || ''));
|
||||||
ai = await visionTool.dispatch(ai, {
|
ai = await visionTool.dispatch(ai, {
|
||||||
images: prepared.images, visionModel: prepared.visionModel, callAI: callAI,
|
images: prepared.images, visionModel: prepared.visionModel, callAI: callAI,
|
||||||
messages: prepared.messages,
|
messages: prepared.messages,
|
||||||
|
|
@ -573,15 +579,22 @@ async function prepareAssistantChat(body) {
|
||||||
var includeContext = body.includeContext !== false;
|
var includeContext = body.includeContext !== false;
|
||||||
var showSources = await showSourcesEnabled();
|
var showSources = await showSourcesEnabled();
|
||||||
|
|
||||||
|
// Phase timings, numbers only, so "is it slow?" can be answered from the
|
||||||
|
// log rather than from a feeling: how long the rewrite took, how long the
|
||||||
|
// library search took, and (in the stream route) when the first token came.
|
||||||
|
var timing = { rewriteMs: 0, searchMs: 0 };
|
||||||
|
var phase = Date.now();
|
||||||
var searchQuery = await rewriteSearchQuery(message, history, chatModel).catch(function(e) {
|
var searchQuery = await rewriteSearchQuery(message, history, chatModel).catch(function(e) {
|
||||||
console.warn('[clinical-assistant] query rewrite skipped:', e.message);
|
console.warn('[clinical-assistant] query rewrite skipped:', e.message);
|
||||||
return message;
|
return message;
|
||||||
});
|
});
|
||||||
|
timing.rewriteMs = Date.now() - phase; phase = Date.now();
|
||||||
var searchResponse = await semanticSearch(searchQuery, {
|
var searchResponse = await semanticSearch(searchQuery, {
|
||||||
limit: searchLimit,
|
limit: searchLimit,
|
||||||
includeContext: includeContext,
|
includeContext: includeContext,
|
||||||
contextChars: contextChars
|
contextChars: contextChars
|
||||||
});
|
});
|
||||||
|
timing.searchMs = Date.now() - phase;
|
||||||
// Retrieval is text-only. The multimodal path called nc_multimodal_search
|
// Retrieval is text-only. The multimodal path called nc_multimodal_search
|
||||||
// against a second hardcoded collection with an embedding service that was
|
// against a second hardcoded collection with an embedding service that was
|
||||||
// never deployed, so it only ever logged "multimodal search skipped".
|
// never deployed, so it only ever logged "multimodal search skipped".
|
||||||
|
|
@ -592,7 +605,7 @@ async function prepareAssistantChat(body) {
|
||||||
var sources = dedupeSources(rawResults).slice(0, searchLimit);
|
var sources = dedupeSources(rawResults).slice(0, searchLimit);
|
||||||
|
|
||||||
if (sources.length === 0) {
|
if (sources.length === 0) {
|
||||||
return { direct: {
|
return { timing: timing, direct: {
|
||||||
success: true,
|
success: true,
|
||||||
answer: 'I could not find a clear match for that question. Please try a more specific clinical term, diagnosis, medication, age group, or textbook topic.',
|
answer: 'I could not find a clear match for that question. Please try a more specific clinical term, diagnosis, medication, age group, or textbook topic.',
|
||||||
sources: [],
|
sources: [],
|
||||||
|
|
@ -603,6 +616,7 @@ async function prepareAssistantChat(body) {
|
||||||
|
|
||||||
var context = formatSourcesForPrompt(sources);
|
var context = formatSourcesForPrompt(sources);
|
||||||
return {
|
return {
|
||||||
|
timing: timing,
|
||||||
message: message,
|
message: message,
|
||||||
showSources: showSources,
|
showSources: showSources,
|
||||||
images: images,
|
images: images,
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue