pediatric-ai-scribe-v3/test/vision-tool.test.js
Daniel fce05a2749
Some checks failed
Forgejo Android APK / Root app tests (push) Successful in 49s
Forgejo Docker Build / Root app tests (push) Successful in 57s
Forgejo Android APK / Build signed APK (push) Successful in 1m58s
Forgejo Docker Build / Build Docker image (push) Successful in 18s
Forgejo Docker Build / Deploy to the host (push) Failing after 0s
fix: the active library view is unmistakable, and a leading question cannot mislead the vision model
Documents and Images were two buttons of the same weight, which reads as two
buttons rather than as a current view and an alternative. The active one now
carries the surface, the weight and a hairline — keyed off aria-selected, so the
visible highlight and what a screen reader announces cannot disagree.

The look_at_image question is written by a model that has not seen the image,
from what the user said, so it can presume something that is not there —
"describe this rash" about a photograph of a drug chart. Answering the
presumption would send it back as fact. The seeing model is now told the
question may assume something absent, and to say so first and describe what is
actually there.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01Dv6sqaY6Vq3ChZHMem3cnU
2026-09-12 16:27:21 +02:00

131 lines
6.4 KiB
JavaScript
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

// A text-only chat model can be lent a pair of eyes: it calls look_at_image, a
// vision model describes the attachment, and the chat model answers from the
// description. The attachment never reaches the model that cannot read it.
const test = require('node:test');
const assert = require('node:assert/strict');
const fs = require('node:fs');
const path = require('node:path');
const visionTool = require('../src/utils/visionTool');
const read = file => fs.readFileSync(path.join(__dirname, '..', file), 'utf8');
const IMAGES = [{ mimeType: 'image/png', dataBase64: 'aW1n' }];
const call = (question) => ({
id: 'call_1', type: 'function',
function: { name: 'look_at_image', arguments: JSON.stringify({ question }) }
});
function recorder(replies) {
const calls = [];
const queue = replies.slice();
return {
calls,
callAI: async (messages, options) => {
calls.push({ messages, options });
return queue.length > 1 ? queue.shift() : queue[0];
}
};
}
test('the tool describes itself as eyesight the model does not have', () => {
const fn = visionTool.tools[0].function;
assert.equal(fn.name, 'look_at_image');
assert.match(fn.description, /You cannot see images yourself/);
assert.equal(fn.parameters.required[0], 'question');
});
test('a call routes the image to the vision model, never to the chat model', async () => {
const r = recorder([{ content: 'A red maculopapular rash over the trunk.' }, { content: 'That pattern fits...' }]);
const out = await visionTool.dispatch({ content: '', toolCalls: [call('describe the rash')] }, {
images: IMAGES, visionModel: 'seeing-model', callAI: r.callAI,
messages: [{ role: 'user', content: 'what is this rash' }], options: { model: 'blind-model' }
});
assert.equal(r.calls.length, 2);
// First hop: the vision model, with the picture.
assert.equal(r.calls[0].options.model, 'seeing-model');
assert.equal(r.calls[0].options.images, IMAGES);
assert.match(r.calls[0].messages[0].content, /describe the rash/);
// Second hop: the chat model, with words and no picture.
assert.equal(r.calls[1].options.model, 'blind-model');
assert.equal(r.calls[1].options.images, undefined, 'the blind model never receives the image');
assert.equal(r.calls[1].options.toolChoice, 'none', 'it has its answer; it must now use it');
const toolMessage = r.calls[1].messages.find(m => m.role === 'tool');
assert.match(toolMessage.content, /maculopapular/);
assert.equal(out.visionHandled, true);
assert.equal(out.content, 'That pattern fits...');
});
test('a question that presumes the wrong thing is corrected, not answered', () => {
// The question is written by a model that has not seen the image, from what
// the user said — so "describe this rash" can arrive about a photograph of a
// drug chart. Answering the presumption would send that back as fact.
const prompt = visionTool.describePrompt('what is the distribution of this rash');
assert.match(prompt, /may assume something the image does not show/);
assert.match(prompt, /If the question presumes something absent/);
assert.match(prompt, /describe what the image does contain/);
});
test('the seeing model is told to report, not to diagnose', () => {
// It has a picture and no conversation, no sources and no system prompt. An
// opinion from it would carry weight it has not earned.
const prompt = visionTool.describePrompt('what is this');
assert.match(prompt, /Report only what is observable/);
assert.match(prompt, /Do not diagnose/);
assert.match(prompt, /say plainly when something is unclear/);
});
test('a reply with no tool call passes straight through', async () => {
const r = recorder([{ content: 'unused' }]);
const ai = { content: 'A direct answer.', toolCalls: [] };
assert.equal(await visionTool.dispatch(ai, { images: IMAGES, visionModel: 'm', callAI: r.callAI }), ai);
assert.equal(r.calls.length, 0);
});
test('another tools call is left for that tool', async () => {
const r = recorder([{ content: 'unused' }]);
const ai = { content: '', toolCalls: [{ id: 'x', type: 'function', function: { name: 'generate_image', arguments: '{}' } }] };
assert.equal(await visionTool.dispatch(ai, { images: IMAGES, visionModel: 'm', callAI: r.callAI }), ai);
assert.equal(r.calls.length, 0);
});
test('calling it with nothing attached, or no vision model, says so plainly', async () => {
const r = recorder([{ content: 'x' }]);
await assert.rejects(
() => visionTool.dispatch({ content: '', toolCalls: [call('what is this')] },
{ images: [], visionModel: 'm', callAI: r.callAI }),
err => /no image attached/i.test(err.message) && err.statusCode === 400);
await assert.rejects(
() => visionTool.dispatch({ content: '', toolCalls: [call('what is this')] },
{ images: IMAGES, visionModel: '', callAI: r.callAI }),
err => /No vision model is configured/.test(err.message) && err.statusCode === 503);
});
test('a vision model that says nothing is an error, not an empty description', async () => {
const r = recorder([{ content: ' ' }]);
await assert.rejects(
() => visionTool.dispatch({ content: '', toolCalls: [call('describe')] },
{ images: IMAGES, visionModel: 'seeing', callAI: r.callAI, messages: [] }),
err => /no description/i.test(err.message));
});
test('the route withholds the attachment only from a model the gateway calls blind', () => {
const route = read('src/routes/clinicalAssistant.js');
assert.match(route, /function assistantToolset\(prepared\)/);
// Delegation is opt-in on an explicit false, never on an unknown.
assert.match(route, /supportsVision\([\s\S]{0,80}\) === false/);
assert.match(route, /tools: imageTool\.tools\.concat\(visionTool\.tools\), images: undefined/);
// Both call sites build the toolset, and both completion paths reuse the same
// decision — a finalize that re-attached the withheld image would undo it.
assert.equal((route.match(/var toolset = assistantToolset\(prepared\);/g) || []).length, 2);
assert.equal((route.match(/images: assistantToolset\(prepared\)\.images/g) || []).length, 2);
});
test('vision capability is asked of the gateway and cached, never guessed from the id', () => {
const util = read('src/utils/modelVision.js');
assert.match(util, /supports_vision === 'boolean' \? info\.supports_vision : null/);
assert.match(util, /never inferred from the model id/);
assert.match(util, /CACHE_MS/);
assert.doesNotMatch(util, /\/gemini|\/gpt-4|test\(modelId\)/, 'no name matching');
});