Files
Stirling-PDF/engine/src/stirling/agents/orchestrator.py
T
ReeceandClaude Opus 4.7 b32a3cf271 Merge origin/main into AI-Form-Fill
Resolution favours main for all conflicts (new orchestrator shape with
artifacts/file_names/conversation_history/resume_with, WorkflowOutcome
enum, RAG, ledger agent, deleted engine/config/.env.example).

Re-wires our three form-fill agents (FormAnalyserAgent, FormFillerAgent,
DocumentExtractorAgent) into the new app layout:
- contracts/__init__.py: adds form-fill exports to __all__
- api/dependencies.py: adds get_form_analyser_agent / get_form_filler_agent / get_document_extractor_agent
- api/app.py: instantiates the three agents in lifespan and registers form_fill_router
- api/routes/__init__.py: exports form_fill_router
- tests/test_stirling_api.py: imports and registers form-fill stubs
- tests/test_stirling_contracts.py: adds back KnowledgeUpdateResponse discriminator test

Form fill remains accessible via its own endpoints (/api/v1/form/ai/analyse,
/fill-batch, /extract). Not wired as an orchestrator delegate — following
main's pattern where orchestrator only routes pdf_edit/pdf_question/user_spec
/math_auditor.

Follow-ups still needed:
- Thread conversation_history into form-fill agent prompts
- Align form-fill response outcomes with WorkflowOutcome enum
- Decide whether engine should return ToolOperationStep plans (main's
  new pattern per #6116) or keep returning fill values directly

All 127 engine tests pass.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-22 13:13:12 +01:00

208 lines
8.7 KiB
Python

from __future__ import annotations
from dataclasses import dataclass
from typing import assert_never
from pydantic_ai import Agent
from pydantic_ai.output import ToolOutput
from pydantic_ai.tools import RunContext
from stirling.agents.document_extractor import DocumentExtractorAgent
from stirling.agents.pdf_edit import PdfEditAgent
from stirling.agents.pdf_questions import PdfQuestionAgent
from stirling.agents.user_spec import UserSpecAgent
from stirling.contracts import (
AgentDraftRequest,
AgentDraftWorkflowResponse,
ExtractedTextArtifact,
OrchestratorRequest,
OrchestratorResponse,
PdfEditRequest,
PdfEditResponse,
PdfQuestionRequest,
PdfQuestionResponse,
SupportedCapability,
ToolOperationStep,
UnsupportedCapabilityResponse,
format_conversation_history,
)
from stirling.contracts.pdf_edit import EditPlanResponse
from stirling.models.agent_tool_models import AgentToolId, MathAuditorAgentParams
from stirling.services import AppRuntime
@dataclass(frozen=True)
class OrchestratorDeps:
runtime: AppRuntime
request: OrchestratorRequest
class OrchestratorAgent:
def __init__(self, runtime: AppRuntime) -> None:
self.runtime = runtime
self.agent = Agent(
model=runtime.fast_model,
output_type=[
ToolOutput(
self.delegate_pdf_edit,
name="delegate_pdf_edit",
description="Delegate requests for PDF modifications and return the PDF edit result.",
),
ToolOutput(
self.delegate_pdf_question,
name="delegate_pdf_question",
description="Delegate questions about PDF contents and return the PDF question result.",
),
ToolOutput(
self.delegate_form_fill,
name="delegate_form_fill",
description="Delegate requests to extract personal information from documents for form filling.",
),
ToolOutput(
self.delegate_user_spec,
name="delegate_user_spec",
description="Delegate requests to create or revise a user agent spec and return the draft result.",
),
ToolOutput(
self.math_auditor_agent,
name="math_auditor_agent",
description=(
"Delegate requests to check arithmetic, validate table totals, "
"audit financial calculations, or verify mathematical accuracy in PDFs."
),
),
ToolOutput(
self.unsupported_capability,
name="unsupported_capability",
description="Return this when none of the delegate outputs fit the request.",
),
],
deps_type=OrchestratorDeps,
system_prompt=(
"You are the top-level orchestrator. "
"Choose exactly one output function that best handles the request. "
"Use delegate_pdf_edit for requested modifications of single or multiple PDFs. "
"Use delegate_pdf_question for questions about PDF contents. "
"Use delegate_user_spec for requests to create or define an agent spec. "
"Use math_auditor_agent for requests to check arithmetic, validate "
"table totals, audit financial calculations, or verify math in PDFs. "
"Use unsupported_capability only when none of the other outputs fit."
),
model_settings=runtime.fast_model_settings,
)
async def handle(self, request: OrchestratorRequest) -> OrchestratorResponse:
if request.resume_with is not None:
return await self._resume(request, request.resume_with)
result = await self.agent.run(
self._build_prompt(request),
deps=OrchestratorDeps(runtime=self.runtime, request=request),
)
return result.output
async def _resume(self, request: OrchestratorRequest, capability: SupportedCapability) -> OrchestratorResponse:
"""Fast-path to get back to the correct endpoint without having to call AI."""
match capability:
case SupportedCapability.PDF_QUESTION:
return await self._run_pdf_question(request)
case SupportedCapability.PDF_EDIT:
return await self._run_pdf_edit(request)
case SupportedCapability.AGENT_DRAFT:
return await self._run_agent_draft(request)
case (
SupportedCapability.ORCHESTRATE
| SupportedCapability.AGENT_REVISE
| SupportedCapability.AGENT_NEXT_ACTION
| SupportedCapability.MATH_AUDITOR_AGENT
):
raise ValueError(f"Cannot resume orchestrator with capability: {capability}")
case _ as unreachable:
assert_never(unreachable)
async def delegate_pdf_edit(self, ctx: RunContext[OrchestratorDeps]) -> PdfEditResponse:
return await self._run_pdf_edit(ctx.deps.request)
async def _run_pdf_edit(self, request: OrchestratorRequest) -> PdfEditResponse:
return await PdfEditAgent(self.runtime).handle(
PdfEditRequest(
user_message=request.user_message,
file_names=request.file_names,
conversation_history=request.conversation_history,
)
)
async def delegate_pdf_question(self, ctx: RunContext[OrchestratorDeps]) -> PdfQuestionResponse:
return await self._run_pdf_question(ctx.deps.request)
async def _run_pdf_question(self, request: OrchestratorRequest) -> PdfQuestionResponse:
extracted_text = self._get_extracted_text_artifact(request)
return await PdfQuestionAgent(self.runtime).handle(
PdfQuestionRequest(
question=request.user_message,
file_names=request.file_names,
page_text=extracted_text.files if extracted_text is not None else [],
conversation_history=request.conversation_history,
)
)
async def delegate_user_spec(self, ctx: RunContext[OrchestratorDeps]) -> AgentDraftWorkflowResponse:
return await self._run_agent_draft(ctx.deps.request)
async def _run_agent_draft(self, request: OrchestratorRequest) -> AgentDraftWorkflowResponse:
return await UserSpecAgent(self.runtime).draft(
AgentDraftRequest(
user_message=request.user_message,
conversation_history=request.conversation_history,
)
)
async def math_auditor_agent(self, ctx: RunContext[OrchestratorDeps]) -> EditPlanResponse:
return EditPlanResponse(
summary="Validate mathematical calculations in the document.",
steps=[
ToolOperationStep(
tool=AgentToolId.MATH_AUDITOR_AGENT,
parameters=MathAuditorAgentParams(),
)
],
)
async def unsupported_capability(
self,
ctx: RunContext[OrchestratorDeps],
capability: str,
message: str,
) -> UnsupportedCapabilityResponse:
return UnsupportedCapabilityResponse(capability=capability, message=message)
def _get_extracted_text_artifact(self, request: OrchestratorRequest) -> ExtractedTextArtifact | None:
for artifact in request.artifacts:
if isinstance(artifact, ExtractedTextArtifact):
return artifact
return None
def _build_prompt(self, request: OrchestratorRequest) -> str:
artifact_summary = self._describe_artifacts(request)
file_names = ", ".join(request.file_names) if request.file_names else "Unknown files"
history = format_conversation_history(request.conversation_history)
return (
f"Conversation history:\n{history}\n"
f"User message: {request.user_message}\n"
f"Files: {file_names}\n"
f"Available artifacts:\n{artifact_summary}"
)
def _describe_artifacts(self, request: OrchestratorRequest) -> str:
if not request.artifacts:
return "- none"
descriptions: list[str] = []
for artifact in request.artifacts:
if isinstance(artifact, ExtractedTextArtifact):
total_pages = sum(len(f.pages) for f in artifact.files)
file_names = [f.file_name for f in artifact.files]
descriptions.append(f"- extracted_text: {total_pages} pages from {file_names}")
continue
descriptions.append("- unknown artifact")
return "\n".join(descriptions)