diff --git a/CHANGELOG.md b/CHANGELOG.md
index 2c8ed0ae..ffa0128c 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -1,6 +1,17 @@
# Changelog
## [Unreleased]
+### Fixed
+
+- **LLM Schema Compliance**: Improved prompts to prevent LLMs from returning objects instead of plain strings for `list[str]` fields
+ - All graph prompts now explicitly state that list fields must contain plain strings only
+ - Added missing `query` and `confidence` fields to search agent output format documentation
+ - Fixes validation errors with less capable models that ignore JSON schema constraints
+- **AG-UI Frontend Types**: Fixed TypeScript interfaces in ag-ui-research example to match backend Python models
+ - `EvaluationResult`: `confidence` → `confidence_score`, `should_continue` → `is_sufficient`, `gaps_identified` → `gaps`, `follow_up_questions` → `new_questions`, added `key_insights`
+ - `ResearchReport`: `question` → `title`, `summary` → `executive_summary`, `findings` → `main_findings`, removed `insights_used`/`methodology`, added `limitations`/`recommendations`/`sources_summary`
+ - Updated Final Report UI to display new fields (Recommendations, Limitations, Sources)
+
## [0.20.1] - 2025-12-11
### Added
diff --git a/examples/ag-ui-research/frontend/components/Agent.tsx b/examples/ag-ui-research/frontend/components/Agent.tsx
index a8dc5f59..8a98a3dd 100644
--- a/examples/ag-ui-research/frontend/components/Agent.tsx
+++ b/examples/ag-ui-research/frontend/components/Agent.tsx
@@ -53,20 +53,22 @@ interface ResearchContext {
}
interface EvaluationResult {
- confidence: number;
+ key_insights: string[];
+ new_questions: string[];
+ gaps: string[];
+ confidence_score: number;
+ is_sufficient: boolean;
reasoning: string;
- should_continue: boolean;
- gaps_identified: string[];
- follow_up_questions: string[];
}
interface ResearchReport {
- question: string;
- summary: string;
- findings: string[];
+ title: string;
+ executive_summary: string;
+ main_findings: string[];
conclusions: string[];
- insights_used: string[];
- methodology: string;
+ limitations: string[];
+ recommendations: string[];
+ sources_summary: string;
}
interface ResearchState {
diff --git a/examples/ag-ui-research/frontend/components/StateDisplay.tsx b/examples/ag-ui-research/frontend/components/StateDisplay.tsx
index b1bb2c39..c0ad3291 100644
--- a/examples/ag-ui-research/frontend/components/StateDisplay.tsx
+++ b/examples/ag-ui-research/frontend/components/StateDisplay.tsx
@@ -58,20 +58,22 @@ interface ResearchContext {
}
interface EvaluationResult {
- confidence: number;
+ key_insights: string[];
+ new_questions: string[];
+ gaps: string[];
+ confidence_score: number;
+ is_sufficient: boolean;
reasoning: string;
- should_continue: boolean;
- gaps_identified: string[];
- follow_up_questions: string[];
}
interface ResearchReport {
- question: string;
- summary: string;
- findings: string[];
+ title: string;
+ executive_summary: string;
+ main_findings: string[];
conclusions: string[];
- insights_used: string[];
- methodology: string;
+ limitations: string[];
+ recommendations: string[];
+ sources_summary: string;
}
interface ResearchState {
@@ -179,7 +181,7 @@ export default function StateDisplay({ state }: StateDisplayProps) {
state.max_iterations > 0
? (state.iterations / state.max_iterations) * 100
: 0;
- const confidence = state.last_eval?.confidence || 0;
+ const confidence = state.last_eval?.confidence_score || 0;
return (
- {state.result.question}
+ {state.result.title}
- Summary
+ Executive Summary
-
+
@@ -1072,7 +1074,7 @@ export default function StateDisplay({ state }: StateDisplayProps) {
marginBottom: "0.5rem",
}}
>
- Key Findings
+ Main Findings
- {state.result.findings.map((finding, idx) => (
+ {state.result.main_findings.map((finding, idx) => (
-
-
-
- Methodology
-
-
-
+ {state.result.recommendations.length > 0 && (
+
+
+ Recommendations
+
+
+ {state.result.recommendations.map((rec, idx) => (
+ -
+
+
+ ))}
+
-
+ )}
+ {state.result.limitations.length > 0 && (
+
+
+ Limitations
+
+
+ {state.result.limitations.map((lim, idx) => (
+ -
+
+
+ ))}
+
+
+ )}
- Insights Used ({state.result.insights_used.length})
+ Sources
- {state.result.insights_used.map((insightId, idx) => {
- const insight = state.context.insights.find(
- (i) => i.id === insightId,
- );
- return (
-
- {insight ? (
-
-
-
- ) : (
-
- Insight ID: {insightId}
-
- )}
-
- );
- })}
+
diff --git a/haiku_rag_slim/haiku/rag/graph/common/prompts.py b/haiku_rag_slim/haiku/rag/graph/common/prompts.py
index 2f17ad7f..ed8d49c2 100644
--- a/haiku_rag_slim/haiku/rag/graph/common/prompts.py
+++ b/haiku_rag_slim/haiku/rag/graph/common/prompts.py
@@ -10,13 +10,17 @@ Responsibilities:
Plan requirements:
- Produce at most 3 sub_questions that together cover the main question.
+- sub_questions must be a list of plain strings, where each string is a complete
+ question. Do NOT use objects with nested fields like {question, details}.
- Each sub_question must be a standalone, self-contained query that can run
without extra context. Include concrete entities, scope, timeframe, and any
qualifiers. Avoid ambiguous pronouns (it/they/this/that).
- Prioritize the highest-value aspects first; avoid redundancy and overlap.
- Prefer questions that are likely answerable from the current knowledge base;
if coverage is uncertain, make scopes narrower and specific.
-- Order sub_questions by execution priority (most valuable first)."""
+- Order sub_questions by execution priority (most valuable first).
+
+Use the gather_context tool once on the main question before planning."""
SEARCH_AGENT_PROMPT = """You are a search and question-answering specialist.
@@ -46,8 +50,13 @@ Each result includes:
- Type: content type like paragraph, table, code, list_item (when available)
- Content: the actual text
-IMPORTANT: In cited_chunks, use the EXACT, COMPLETE chunk ID (the full UUID).
-Do NOT truncate or shorten chunk IDs.
+Output format:
+- query: Echo the question you are answering
+- answer: Your concise answer based on the retrieved content
+- cited_chunks: List of plain strings containing only the chunk UUIDs (not objects)
+- confidence: A score from 0.0 to 1.0 indicating answer confidence
+
+IMPORTANT: Use the EXACT, COMPLETE chunk ID (full UUID). Do NOT truncate IDs.
Guidelines:
- Base answers strictly on retrieved content - do not use external knowledge.
diff --git a/haiku_rag_slim/haiku/rag/graph/deep_qa/prompts.py b/haiku_rag_slim/haiku/rag/graph/deep_qa/prompts.py
index 56d4d4d9..9dc46615 100644
--- a/haiku_rag_slim/haiku/rag/graph/deep_qa/prompts.py
+++ b/haiku_rag_slim/haiku/rag/graph/deep_qa/prompts.py
@@ -9,8 +9,10 @@ Task:
- Be clear, accurate, and well-structured
Output format:
+- query: Echo the original question being answered
- answer: The complete answer to the original question (2-4 paragraphs)
-- cited_chunks: List of chunk IDs (from sub-answers) that directly support your answer
+- cited_chunks: List of plain strings containing chunk IDs (UUIDs only, not objects)
+- confidence: A score from 0.0 to 1.0 indicating answer confidence
Guidelines:
- Start directly with the answer - no preamble like "Based on the research..."
@@ -30,7 +32,7 @@ Task:
Output format:
- is_sufficient: Boolean indicating if we can answer the question comprehensively
- reasoning: Clear explanation of your assessment
-- new_questions: List of specific follow-up questions needed (empty if sufficient)
+- new_questions: List of plain strings, each a specific follow-up question (not objects)
Guidelines:
- Be strict but reasonable in your assessment
diff --git a/haiku_rag_slim/haiku/rag/graph/research/prompts.py b/haiku_rag_slim/haiku/rag/graph/research/prompts.py
index 540019ee..70dfbb22 100644
--- a/haiku_rag_slim/haiku/rag/graph/research/prompts.py
+++ b/haiku_rag_slim/haiku/rag/graph/research/prompts.py
@@ -16,19 +16,20 @@ Tasks:
Output format (map directly to fields):
- highlights: list of insights with fields {summary, status, supporting_sources,
originating_questions, notes}. Use status one of {validated, open, tentative}.
+ supporting_sources and originating_questions must be lists of plain strings.
- gap_assessments: list of gaps with fields {description, severity, blocking,
resolved, resolved_by, supporting_sources, notes}. Severity must be one of
- {low, medium, high}. resolved_by may reference related insight summaries if no
- stable identifier yet.
-- resolved_gaps: list of identifiers or descriptions for gaps now closed.
-- new_questions: up to 3 standalone, specific sub-questions (no duplicates with
- existing ones).
+ {low, medium, high}. resolved_by and supporting_sources must be lists of plain strings.
+- resolved_gaps: list of plain strings (identifiers or descriptions for gaps now closed).
+- new_questions: list of plain strings, up to 3 standalone questions (no duplicates).
- commentary: 1–3 sentences summarizing what changed this round.
+All list fields must contain plain strings only, not objects.
+
Guidance:
- Be concise and avoid repeating previously recorded information unless it
changed materially.
-- Tie supporting_sources to the evidence used; omit if unavailable.
+- For supporting_sources, use only the document_uri strings from the sources.
- Only propose new sub_questions that directly address remaining gaps.
- When marking a gap as resolved, ensure the rationale is clear via
resolved_by or notes."""
@@ -58,15 +59,15 @@ Strictness:
- Treat unresolved high-severity or blocking gaps as a hard stop.
Output fields must line up with EvaluationResult:
-- key_insights: concise bullet-ready statements of the most decision-relevant
- insights (cite status if helpful).
-- new_questions: follow-up sub-questions (max 3) meeting the specificity rules.
-- gaps: list remaining blockers; reuse wording from the tracked gaps when
- possible to aid downstream reconciliation.
+- key_insights: list of plain strings, concise bullet-ready statements.
+- new_questions: list of plain strings, follow-up sub-questions (max 3).
+- gaps: list of plain strings, remaining blockers (reuse wording from tracked gaps).
- confidence_score: numeric in [0,1].
- is_sufficient: true only when no blocking gaps remain.
- reasoning: short narrative tying the decision to evidence coverage.
+All list fields must contain plain strings only, not objects.
+
Remember: prefer maintaining continuity with the structured context over
introducing new terminology."""
@@ -82,16 +83,13 @@ Goals:
Report guidelines (map to output fields):
- title: concise (5–12 words), informative.
- executive_summary: 3–5 sentences summarizing the overall answer.
-- main_findings: 4–8 one‑sentence bullets; each reflects evidence from the
- research (do not include inline citations or snippet text).
-- conclusions: 2–4 bullets that follow logically from findings.
-- recommendations: 2–5 actionable bullets tied to findings.
-- limitations: 1–3 bullets describing key constraints or uncertainties.
-- sources_summary: List specific sources used with document paths, page numbers,
- and section headings where available. Format each as:
- "- /path/to/document.pdf (p. 5, Section: Introduction)" or
- "- /path/to/file.md (Section: Getting Started)"
- Include one bullet per distinct source document.
+- main_findings: list of plain strings, 4–8 one‑sentence bullets reflecting evidence.
+- conclusions: list of plain strings, 2–4 bullets following logically from findings.
+- recommendations: list of plain strings, 2–5 actionable bullets tied to findings.
+- limitations: list of plain strings, 1–3 bullets describing constraints or uncertainties.
+- sources_summary: single string listing sources with document paths and page numbers.
+
+All list fields must contain plain strings only, not objects.
Style:
- Base all content solely on the collected evidence.