Documentation organization

This commit is contained in:
2026-08-27 10:28:23 -03:00
parent 472d44074c
commit 42ab000c7b
15 changed files with 6548 additions and 7060 deletions

117
.idea/workspace.xml generated
View File

@@ -6,120 +6,7 @@
<component name="ChangeListManager">
<list default="true" id="30a0e1d8-9d7d-469b-b241-f300911cee8a" name="Changes" comment="Ajustes na documentação e remanejamento dos folders">
<change beforePath="$PROJECT_DIR$/.idea/workspace.xml" beforeDir="false" afterPath="$PROJECT_DIR$/.idea/workspace.xml" afterDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/.oca/custom_code_review_guidelines.txt" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/Arquitetura_Geral_Agent_Framework_OCI.docx" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/FIX_DEADLOCK_SEQUENCE_CROSS_LOOP.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/IMPLEMENTACAO_WORKFLOWS_TRANSACIONAIS.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/INVENTARIO_AGENT_GATEWAY_MCP_GATEWAY.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/Implementando_Basic_Auth.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/Long_Term_Memory_Implementation_Guide_EN.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/MANUAL_AGENT_PLATFORM_GATEWAYS.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/MANUAL_EXECUCAO_AGENT_GATEWAY_MCP_GATEWAY_FRONTEND.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/MCP_GATEWAY_RUNBOOK.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/Manual de Roteamento Multi-Agent.docx" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/Manual_Desenvolvedor_AI_Agent_Framework_OCI.docx" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/Manual_Integracao_MCP_Servers_Agent_Framework.docx" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/Manual_Long_Term_Memory_PT.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_AGENT_GATEWAY_AND_MCP_GATEWAY_EVOLUTION.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_CHECKPOINT_ENTERPRISE.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_ENTERPRISE_ROUTING.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_FIRST_ENTERPRISE_DELTA.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_FIRST_ENTERPRISE_PLUS.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_FIRST_MAX_OPERATIONAL_FIXES.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_FIRST_READY.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_GUARDRAILS_IMPLEMENTADOS.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_MAX_OPERACIONAL.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_MCP.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_MULTI_AGENT_ISOLATION.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_ROUTE_STICKINESS_SEMANTICA.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_ROUTING_MODES.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_SEMANTIC_ROUTE_STICKINESS.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_TEMPLATE_BUSINESS_CONTEXT_V2.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_TESTES_UNITARIOS.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_TOOL_POLICIES.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_old.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/README_old2.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/RELEASE_NOTES_GENERIC_DETERMINISTIC_INTENT_SHIFT_V15.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/RELEASE_NOTES_MCP_PARAMETER_EXTRACTION_FIX.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/RELEASE_NOTES_ROUTE_STICKINESS_DETERMINISTIC_INTENT_SHIFT_V14.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/RELEASE_NOTES_ROUTE_STICKINESS_TRANSACTION_SHIFT.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/RELEASE_NOTES_TOOL_POLICIES.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/Release Notes/DIFF_AGENT_FRAMEWORK_LOCAL_VS_OCI_2026-08-12.pdf" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/Route_Stickiness_Semantica_Agent_Framework_OCI.docx" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/TEST_RESULTS_ROUTE_STICKINESS.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/VALIDACAO_TRANSACIONAL_BACKEND_MCP.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/docs_GLOBAL_SUPERVISOR_VALIDATION.txt" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/docs_VALIDATION_GUARDRAILS_IC.txt" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/Documentacao/img.png" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/README.md" beforeDir="false" afterPath="$PROJECT_DIR$/README.md" afterDir="false" />
<change beforePath="$PROJECT_DIR$/README_en.md" beforeDir="false" afterPath="$PROJECT_DIR$/README_en.md" afterDir="false" />
<change beforePath="$PROJECT_DIR$/docs/01_billing_agent_invoice_policy.pdf" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/02_orders_agent_lifecycle_policy.pdf" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/03_product_agent_catalog_policy.pdf" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/04_support_agent_sla_policy.pdf" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/05_business_context_rag_flow.pdf" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/ADR_TRANSACTIONAL_WORKFLOW_ENGINE.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/EXTERNAL_GUARDRAILS_JUDGES.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/JUDGES_TRANSACTIONAL_SAMPLING_FIX.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/LLM_RICH_RESPONSE.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/MCP_GATEWAY_DISCOVERY.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/MODULAR_REMAP.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/PERFORMANCE_OPTIMIZATIONS_MCP_JUDGES_RAG.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/RAG_PROVIDER_KBDB.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/README_rag_samples.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/TRANSACTION_OPERATIONAL_EVIDENCE_FIX.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/TRANSACTION_STATE_DEVELOPER_GUIDE.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/TRANSACTION_STATE_DEVELOPER_GUIDE_en.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/docs_GLOBAL_SUPERVISOR_VALIDATION.txt" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/docs_VALIDATION_GUARDRAILS_IC.txt" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/01_authentication.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/02_deterministic_transactional_workflow.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/03_domain_requested_llm_composition.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/04_domain_requested_rag.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/05_long_term_memory.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/06_offline_workflow_regression.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/07_pause_resume_workflow.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/08_route_stickiness.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/09_voice_interruption_replay.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/10_workflow_error_recovery.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/11_clarification.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/12_durable_idempotency.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/13_dynamic_transaction_states.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/14_post_finalization_replay.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/15_retrieval_tool_guardrails.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/README.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/01_authentication.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/02_deterministic_transactional_workflow.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/03_domain_requested_llm_composition.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/04_domain_requested_rag.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/05_long_term_memory.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/06_offline_workflow_regression.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/07_pause_resume_workflow.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/08_route_stickiness.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/09_voice_interruption_replay.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/10_workflow_error_recovery.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/11_clarification.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/12_durable_idempotency.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/13_dynamic_transaction_states.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/14_post_finalization_replay.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/15_retrieval_tool_guardrails.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/en/README.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/01_authentication.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/02_deterministic_transactional_workflow.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/03_domain_requested_llm_composition.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/04_domain_requested_rag.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/05_long_term_memory.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/06_offline_workflow_regression.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/07_pause_resume_workflow.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/08_route_stickiness.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/09_voice_interruption_replay.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/10_workflow_error_recovery.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/11_clarification.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/12_durable_idempotency.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/13_dynamic_transaction_states.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/14_post_finalization_replay.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/15_retrieval_tool_guardrails.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/features/pt-BR/README.md" beforeDir="false" />
<change beforePath="$PROJECT_DIR$/docs/developer/pt/02_routing_stickiness_and_intent_shift.md" beforeDir="false" afterPath="$PROJECT_DIR$/docs/developer/pt/02_routing_stickiness_and_intent_shift.md" afterDir="false" />
</list>
<option name="SHOW_DIALOG" value="false" />
<option name="HIGHLIGHT_CONFLICTS" value="true" />
@@ -209,7 +96,7 @@
<workItem from="1785414225783" duration="148000" />
<workItem from="1785414447653" duration="704000" />
<workItem from="1785630146329" duration="316000" />
<workItem from="1787832640995" duration="1220000" />
<workItem from="1787832640995" duration="3537000" />
</task>
<task id="LOCAL-00001" summary="Ajustes na documentação e remanejamento dos folders">
<option name="closed" value="true" />

View File

@@ -1,91 +1,92 @@
### Agent Framework OCI Architecture and Concepts
### Purpose of this document
This document **does not replace the root `README_en.md`** and does not duplicate the end-to-end agent development tutorial.
This document **does not replace the root `README_en.md`** and does not repeat the agent-creation tutorial.
Use:
- [`README_en.md`](../../../README_en.md) to develop, configure, run and test an agent end to end;
- this document to understand architecture, responsibility boundaries, components and where each implementation belongs;
- the other manuals in this folder to deepen a specific capability or troubleshoot a problem.
- [`README_en.md`](../../../README_en.md) to develop, configure, run, and test an agent end to end;
- this document to understand the architecture, responsibility boundaries, components, and where each type of implementation belongs;
- the other manuals in this folder to deepen a specific capability or solve a problem.
The separation is intentional: there is **one main tutorial** and multiple **specialized reference manuals**.
The separation is intentional: there is **one main tutorial** and several **specialized reference manuals**.
### Source of truth
When documentation differs, use this order:
When documentation diverges, use this order:
1. code for the version in use;
2. `README.md` / `README_en.md` for the same version;
2. `README.md` / `README_en.md` from the same version;
3. normative SPECs/SDDs;
4. specialized manuals in this folder;
5. release notes and `README_old*` only as historical material.
5. release notes and `README_old*` only as history.
### Platform mental model
Agent Framework OCI is a layered platform.
Agent Framework OCI should be understood as a layered platform.
The **framework core** provides reusable, domain-neutral mechanisms: runtime, state, memory, routing, tool integration, guardrails, judges, persistence, observability and common contracts.
The **framework core** provides reusable, domain-neutral mechanisms: runtime, state, memory, routing, tool integration, guardrails, judges, persistence, observability, and common contracts.
The **agent** contains use-case-specific behavior: intents, prompts, domain rules, agent-specific policies, business workflows, mappings, integrations and external components owned by that agent.
The **agent** contains what is specific to the use case: intents, prompts, domain rules, specific policies, business workflow, mappings, integrations, and external components that belong to that agent.
**Gateways** handle cross-cutting ingress, governance and integration concerns. They should not absorb agent business logic.
**Gateways** handle cross-cutting ingress, governance, and integration responsibilities. They should not absorb the agent's business logic.
**MCP Servers** encapsulate tools and integrations with domain or legacy services. The **MCP Gateway** provides centralized tool catalog and governance.
**MCP Servers** encapsulate tools and integrations with domain or legacy services. The **MCP Gateway** provides centralized catalog and governance for these tools.
### Main components
| Component | Primary responsibility | Must not contain |
| Component | Main responsibility | Must not contain |
|---|---|---|
| `libs/agent_framework/` | Generic runtime, contracts, state, memory, routing, guardrails, judges and common integrations | Company- or agent-specific business rules |
| `templates/agent_template_backend/` | Executable reference for creating agents | A permanent fork of the core |
| `apps/agent_gateway/` | Governed ingress, cross-cutting policies, rate limits, auth and metadata | Business workflow |
| `apps/channel_gateway/` | Adapt channels to canonical contracts | Agent business logic |
| `apps/mcp_gateway/` | Central tool catalog, authorization and execution | Conversational orchestration |
| `mcp/servers/` | Domain tools and integrations | Global agent orchestration |
| `evals/` | Certification and regression | Production business logic |
| `deploy/` | Containers and Kubernetes artifacts | Functional rules |
| `libs/agent_framework/` | Generic runtime, contracts, state, memory, routing, guardrails, judges, common integrations | Rule specific to a company or agent |
| `templates/agent_template_backend/` | Executable reference for creating agents | Permanent fork of the core |
| `apps/agent_gateway/` | Governed ingress, cross-cutting policies, rate limit, authentication, metadata | Business workflow |
| `apps/channel_gateway/` | Channel adaptation to the canonical contract | Agent business rule |
| `apps/mcp_gateway/` | Catalog, authorization, and centralized tool execution | Conversational logic |
| `mcp/servers/` | Integrations and tools by domain | Global agent orchestration |
| `evals/` | Certification and regression | Production logic |
| `deploy/` | Containers and Kubernetes | Functional rules |
### Conceptual request flow
A typical request goes through the following responsibilities:
```text
Channel
Canal
|
v
Channel Gateway
|
v
Agent Gateway
| governance / auth / rate limit / metadata
| governança / autenticação / rate limit / metadata
v
Agent backend
Backend do agente
|
+--> Routing / stickiness / intent
|
+--> State / memory / checkpoint
+--> Estado / memória / checkpoint
|
+--> Guardrails / judges
|
+--> Workflow / transaction policies
+--> Workflow / políticas transacionais
|
+--> MCP Gateway
|
+--> MCP Server A --> legacy system
+--> MCP Server B --> external service
+--> MCP Server C --> domain API
+--> MCP Server A --> sistema legado
+--> MCP Server B --> serviço externo
+--> MCP Server C --> API de domínio
```
Not every deployment must use every component. Composition follows agent needs and platform contracts.
Not every deployment needs to use all components. Composition should follow the agent's needs and the platform contracts.
### Agent runtime
The current runtime is based on `AgentRuntimeMixin` and `RuntimeContext`.
The template imports runtime through `app.agents.runtime`, which re-exports the official framework implementation. This prevents each agent from maintaining a divergent copy.
The template imports the runtime through `app.agents.runtime`, which re-exports the framework's official implementation. The goal is to prevent each agent from maintaining its own divergent copy of the runtime.
Current APIs confirmed in this version include:
Current APIs confirmed in the code include:
```python
AgentRuntimeMixin.get_runtime_context()
@@ -100,11 +101,11 @@ AgentRuntimeMixin.transaction_confirmation_message()
AgentRuntimeMixin.build_direct_mcp_answer()
```
Developers should prefer these runtime capabilities instead of rebuilding equivalent logic inside each agent.
These APIs represent runtime capabilities. Developers should prefer them over manually rebuilding the same logic inside each agent.
### Configuration versus code
A core framework principle is to keep selectable behavior in configuration.
A central framework guideline is that configurable behavior should remain in configuration.
Examples:
@@ -113,130 +114,151 @@ Examples:
- tools: `config/tools.yaml`;
- MCP Servers and mappings: corresponding MCP configuration;
- LLM profiles: `llm_profiles.yaml`;
- policies and extensions: capability-specific configuration.
- policies and extensions: capability-specific configuration files.
Code implements mechanisms. YAML/config selects behavior whenever this can be done without weakening safety or contracts.
Code should implement mechanisms. YAML/config should select behavior whenever that can be done without compromising security or contracts.
### Framework versus agent responsibility
### Separation between framework and agent
A change belongs to the **framework** when it introduces a mechanism reusable by multiple agents.
A change belongs to the **framework** when it introduces a mechanism reusable by different agents.
A change belongs to the **agent** when it expresses company/domain behavior.
Examples:
If the core must import a concrete agent module to work, that boundary is probably broken.
- new guardrail SPI;
- new rich LLM response contract;
- new generic checkpoint capability;
- new configurable tool-policy mechanism;
- new generic routing strategy.
### State, memory and checkpoint are different concepts
A change belongs to the **agent** when it expresses a rule from a domain or company.
**Execution state** represents what is happening in the turn/workflow.
Examples:
- which charges can be disputed;
- a telecom-specific prompt;
- VAS rules;
- internal company codes;
- legacy-service mapping;
- specific phraseology.
If the core needs to import a concrete agent module in order to work, this separation has probably been broken.
### State, memory, and checkpoint are different concepts
**Execution state** represents what is happening in the turn and workflow.
**Conversation memory** preserves conversational context.
**Long-Term Memory** stores durable facts associated with business identity.
**Long-Term Memory** stores durable facts associated with a business identity.
**Checkpointing** persists LangGraph state snapshots for resume.
**Checkpoint** persists LangGraph state snapshots for resume.
An old checkpoint alone must not determine which transaction is active. Functional decisions should use canonical transaction state.
An old checkpoint must not, by itself, determine which transaction is active. The functional decision must use canonical transaction state.
### Routing and execution are separate responsibilities
### Routing and execution are different responsibilities
Routing answers: **which agent/intent should handle the message?**
Routing answers: **which agent/intent should handle this message?**
Execution answers: **what should that agent do now?**
Route stickiness preserves continuity but must not block an explicit intent change. During a transaction, expected parameters and valid confirmation have precedence to avoid false intent shifts.
Route stickiness preserves continuity, but it must not prevent an explicit intent change. During a transaction, expected parameters and valid confirmation take precedence to avoid false intent shifts.
See [Routing, Stickiness and Intent Shift](./02_routing_stickiness_and_intent_shift.md).
Full details: [Routing, Stickiness, and Intent Shift](./02_routing_stickiness_and_intent_shift.md).
### Tools and MCP
A tool is an invokable capability.
A tool represents an invokable capability.
An MCP Server implements or exposes that capability.
The MCP Server implements or exposes that capability.
The MCP Gateway organizes catalog, authorization, mapping and centralized execution.
The MCP Gateway organizes catalog, authorization, mapping, and centralized execution.
The agent decides **when** a tool is needed; MCP determines **how** the corresponding service is accessed.
The agent decides **when** a tool should be used in its flow; the tool/MCP decides **how** to access the corresponding service.
See [MCP, Tools, Policies and Parameter Extraction](./04_mcp_integration_tools_and_policies.md).
Full details: [MCP, Tools, Policies, and Parameter Extraction](./04_mcp_integration_tools_and_policies.md).
### Transactions
Side-effecting operations require different handling from read-only queries.
Operations with side effects require different handling from queries.
The framework provides state, confirmation, policy and deterministic workflow mechanisms. Concrete domain rules remain in the agent.
The framework provides state, confirmation, policy, and deterministic-workflow mechanisms. Concrete rules remain in the agent.
An LLM may participate in interpretation and composition, but it must not be the sole source of truth for claiming that a critical operation was executed.
The LLM may participate in interpretation and composition, but it must not be the only source of truth for claiming that a critical operation was executed.
See [Transactional Workflows and State](./03_transaction_workflows_and_state.md).
Full details: [Transactional Workflows and State](./03_transaction_workflows_and_state.md).
### Guardrails and judges
### Guardrails and Judges
The core provides native mechanisms and extension points. Domain-specific guardrails/judges belong to the agent and should be loaded through configuration rather than hardcoded imports inside the core.
Guardrails control or validate behavior during processing.
See [Guardrails, Judges and Transaction Evaluation](./06_guardrails_judges_and_transaction_evaluation.md).
Judges evaluate quality, grounding, and other criteria.
### RAG, memory and tools are not interchangeable
The core provides native mechanisms and extension points. Domain-specific guardrails/judges should be loaded by the agent through configuration, avoiding specific imports inside the framework.
Full details: [Guardrails, Judges, and Transaction Evaluation](./06_guardrails_judges_and_transaction_evaluation.md).
### RAG, memory, and tools are not equivalent
- **RAG** retrieves knowledge.
- **Memory** preserves context/facts.
- **Tools** query or execute external capabilities.
- **Tool** executes or queries an external capability.
Using the wrong mechanism creates difficult-to-diagnose behavior.
Choosing the wrong mechanism creates bugs that are difficult to diagnose. Information that needs to be updated in a system should not be solved only through RAG; a durable customer fact should not depend only on prompt history.
### Observability as a cross-cutting contract
Routing, agent, transaction, tool, guardrail, judge and failure events must be correlatable.
Routing, agent, transaction, tool, guardrail, judge, and failure must be correlatable.
Observability records what happened; it must not become business-state control.
Observability should record what happened, but it must not control business state. Sequence, trace IDs, and labels are diagnostic and audit infrastructure.
See [Observability, Persistence and Operational Readiness](./11_observability_persistence_and_operational_readiness.md).
Full details: [Observability, Persistence, and Operational Readiness](./11_observability_persistence_and_operational_readiness.md).
### Where a new feature belongs
### Where to place a new feature
Before implementing a feature, ask:
Before implementing, ask these questions:
1. Is it reusable by multiple agents?
2. Does it contain domain-specific rules?
3. Does it require state across turns?
1. Is the capability reusable by different agents?
2. Is there a domain-specific rule?
3. Does it need state across turns?
4. Does it produce side effects?
5. Does it depend on an external system?
6. Should it be configurable?
7. Must it be observable?
8. Must a guardrail/judge evaluate it?
7. Does it need to appear in observability?
8. Does it need to be evaluated by a guardrail/judge?
A reusable capability normally starts in the core and is enabled/configured by the agent. A business rule normally starts in the agent and uses core interfaces.
A reusable feature normally starts in the core and is enabled/configured by the agent. A business rule normally starts in the agent and uses core interfaces.
### Anti-patterns
Avoid:
- importing a concrete agent package inside the core;
- duplicating `AgentRuntimeMixin` per agent;
- hardcoding agent, intent, tool or company names in runtime;
- treating LLM output as proof of operation execution;
- treating an old checkpoint as the active transaction;
- executing transactional operations without required policy/confirmation;
- directly coupling agents to many services when MCP Gateway is the intended layer;
- duplicating `AgentRuntimeMixin` in every agent;
- hardcoding agent, intent, tool, or company names in the runtime;
- using an LLM response as proof that an operation was executed;
- confusing an old checkpoint with the active transaction;
- executing a transactional operation without policy/confirmation when it is required;
- coupling an agent directly to dozens of services when MCP Gateway is the intended layer;
- creating a new functional document for every bug fix instead of updating the feature manual.
### Recommended path for a new developer
1. Read this architecture overview.
2. Follow [`README_en.md`](../../../README_en.md) end to end.
3. Use the specialized manual when reaching a specific capability.
4. For failures, start from the [Developer Index](./INDEX_DEVELOPER_GUIDE.md), under **Search by problem**.
5. Before copying historical code, confirm the API/import in the current template and core.
1. Read the architectural overview in this document.
2. Follow [`README_en.md`](../../../README_en.md) from beginning to end to create and run an agent.
3. When you reach a specific capability, use the corresponding specialized manual.
4. For failures, start with the [Developer Index](./INDEX_DEVELOPER_GUIDE.md), in the **Search by problem** section.
5. Before copying old code, confirm the API/import in the current template and core.
### Related documents
- [Main tutorial — README_en.md](../../../README_en.md)
- [Routing, Stickiness and Intent Shift](./02_routing_stickiness_and_intent_shift.md)
- [Main tutorial — README.md](../../../README.md)
- [Routing, Stickiness, and Intent Shift](./02_routing_stickiness_and_intent_shift.md)
- [Transactional Workflows and State](./03_transaction_workflows_and_state.md)
- [MCP, Tools, Policies and Parameters](./04_mcp_integration_tools_and_policies.md)
- [MCP, Tools, Policies, and Parameters](./04_mcp_integration_tools_and_policies.md)
- [Gateways and Authentication](./05_agent_gateway_mcp_gateway_and_auth.md)
- [Guardrails and Judges](./06_guardrails_judges_and_transaction_evaluation.md)
- [RAG and BusinessContext](./07_rag_business_context_and_grounding.md)
- [Long-Term Memory and Checkpoint](./08_long_term_memory_and_checkpoint.md)
- [LLM Rich Response](./09_llm_rich_response_reasoning.md)
- [Performance, Cache and Async Runtime](./10_performance_cache_and_async_runtime.md)
- [Performance, Cache, and Async Runtime](./10_performance_cache_and_async_runtime.md)
- [Observability and Operational Readiness](./11_observability_persistence_and_operational_readiness.md)

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@@ -1,73 +1,102 @@
### Guardrails, Judges and Transaction Evaluation
### Guardrails, Judges, and Transaction Evaluation
### How to use this manual
This is a **specialized reference manual**. It does not replace the main tutorial.
- To build an agent end to end, use [`README_en.md`](../../../README_en.md).
- Use this document when implementing, deep-diving or troubleshooting **native/external guardrails, judges, transactional sampling and grounding**.
- Historical examples consolidated here must be interpreted against the current framework API.
- If documentation differs, the current code and root README take precedence.
- To create an agent from start to finish, use [`README_en.md`](../../../README_en.md).
- Use this document when you need to implement, deepen, or diagnose **native/external guardrails, judges, transactional sampling, and grounding**.
- Historical examples consolidated here should be read in light of the framework's current API.
- In case of divergence, the code for the version and the current `README_en.md` take precedence.
### Relationship with the main tutorial
`README_en.md` introduces this capability as part of the normal development flow. This manual consolidates details previously spread across `docs/`, `Documentacao/`, release notes, validation records and specialized guides.
The `README_en.md` presents this capability in the normal development flow. This manual brings together details that were distributed across `docs/`, `Documentacao/`, release notes, validations, and specialized guides.
Its purpose is to answer **“how does this feature work in depth and how do I troubleshoot it?”** without becoming a second copy of the main tutorial.
The goal here is to answer **“how does this feature work in depth and how do I solve problems with it?”**, without turning this file into a second copy of the main tutorial.
### Scope
Native/external guardrails, judges, transactional sampling and grounding.
Native/external guardrails, judges, transactional sampling, and grounding.
### Consolidated technical content
### Guardrails, Judges and Transaction Evaluation
### Guardrails, Judges, and Transaction Evaluation
This guide explains validation layers and how agent-specific policies extend the framework without introducing domain coupling.
Manual for input/output guardrails, agent-specific extensions, external judges, mandatory execution on transactions, and the signals/evidence used during evaluation.
### Guardrail stages
### How to use this document
Input guardrails validate/sanitize/block user input before domain execution. Output guardrails validate the produced response before it leaves the runtime. Optional rails can be enabled according to agent/environment policy.
This is the consolidated development document for this subject. It brings together architecture, configuration, examples, runtime behavior, compatibility, tests, and troubleshooting that were previously distributed across several files. Source sections were preserved when they provided distinct technical details; release notes were incorporated as current behavior or correction history.
### Agent-owned extensions
### Guardrails implemented in the framework
The framework exposes an SPI/configuration model for external guardrails and judges. An agent points configuration to implementation classes in its own package. The shared framework must not import concrete telecom, retail or company validation modules.
> Content consolidated from `Documentacao/README_GUARDRAILS_IMPLEMENTADOS.md`.
Synchronous validators may execute in worker threads; asynchronous validators execute on the event loop. Independent judges may execute concurrently to reduce latency while the configured logical result order is preserved.
This version adds a pragmatic guardrail layer to `agent_framework`, inspired by separating rails by stage: input, output, retrieval, and execution/tool.
### Transactional judge sampling
### Input rails
Normal evaluation may use sampling, but transactional interactions can be configured with `always_run_for_transactional`. Transaction detection occurs before applying `sample_rate` so critical side-effecting paths are not randomly skipped.
- `MSIZE` — blocks excessively large messages.
- `MSK` — masks CPF, CNPJ, phone, e-mail, card, postal code, RG, tokens, and keys.
- `TOX` — detects toxicity and records severity without blocking by default.
- `PINJ` — detects prompt injection and records a score.
- `JBRK` — detects jailbreak/bypass roleplay and records a score.
- `VLOOP` — blocks repetitive conversational loops.
Signals may include transaction lifecycle state, required/received confirmation, selected or pending tool call, tool-policy result and MCP execution results. Detection intentionally uses multiple signals instead of depending on a single field.
### Output rails
### Operational evidence
- `PII_OUT` — masks PII in the agent response.
- `CMP` — softens absolute promises and excessive guarantee language.
- `REVPREC` — blocks verbalization of an operational action without tool confirmation.
- `GND` — signals grounding/risk when there is a specific answer without evidence.
- `ALUC_RISK` — marks hallucination risk for telemetry and judges.
Judges must distinguish a model claim from an executed action. MCP results and transaction evidence provide grounding for assertions such as cancellation, credit, update or protocol creation.
### Optional rails
### Compatibility
- `RET_REL` — validates retrieval-chunk relevance using a minimum score.
- `TOOL_VAL` — validates MCP/tool name, required arguments, negative values, and allowlist.
Legacy validators may use temporary compatibility shims during migration, but new code should depend on the external SPI/configuration. Native framework guardrails continue to coexist with agent-specific policies.
### Files changed
### Testing
- `agent_framework/src/agent_framework/guardrails/rails.py`
- `agent_framework/src/agent_framework/guardrails/pipeline.py`
- `agent_framework/src/agent_framework/guardrails/__init__.py`
Test allow/sanitize/block behavior, exceptions/fail-closed behavior where configured, sync/async external validators, judge concurrency, transactional sample-rate bypass, MCP evidence propagation and isolation between two agents with different policies.
### Quick use
### Source material consolidated
```python
from agent_framework.guardrails.pipeline import GuardrailPipeline
- `Documentacao/README_GUARDRAILS_IMPLEMENTADOS.md`
- `docs/EXTERNAL_GUARDRAILS_JUDGES.md`
- `docs/JUDGES_TRANSACTIONAL_SAMPLING_FIX.md`
- Global Supervisor and guardrail validation records under `docs/`
pipeline = GuardrailPipeline()
### Detailed normative and implementation reference
sanitized_input, input_decisions = await pipeline.run_input(
user_text,
{"history_texts": history_texts},
)
The sections below preserve the detailed English project specifications and implementation guides relevant to this capability. They are included here so a developer does not need to reconstruct the behavior from separate documents.
final_answer, output_decisions = await pipeline.run_output(
answer,
context,
)
```
### External guardrails and judges SPI
For tools/MCP:
> Consolidated from `docs/EXTERNAL_GUARDRAILS_JUDGES.md`.
```python
_, decisions = await pipeline.run_tool(
"cancelar_produto",
{"produto": "VAS", "valor": 0},
{
"required_args": ["produto"],
"allowed_tools": ["cancelar_produto", "consultar_fatura"],
},
)
```
### SPI for external guardrails and judges
> Content consolidated from `docs/EXTERNAL_GUARDRAILS_JUDGES.md`.
`agent_framework_oci` supports agent-owned guardrails and judges without importing domain code into the core.
@@ -88,602 +117,107 @@ judges:
Native entries remain unchanged. External synchronous `evaluate()` methods execute in worker threads via `asyncio.to_thread`; asynchronous methods execute concurrently on the framework event loop. Judges run concurrently with `asyncio.gather`, preserving YAML result order. Agent plugins should reuse the LLM supplied by the framework rather than instantiate a separate provider.
The core must not reference a concrete agent package, company, product, telecom identifier or domain-specific policy. Domain-specific variants belong to the agent and should receive distinct public codes/names.
The core must not reference a concrete agent package, company, product, telecom identifier, or domain-specific policy. Domain-specific variants belong to the agent and should receive distinct public codes/names.
### Compatibility rule
Domain policies must not be replaced by cosmetically generic text inside the core while losing the original policy. The generic core implementation and the agent-specific implementation may coexist; the embedding agent explicitly selects its own code/name in YAML.
Legacy business validators should migrate to the agent domain. A temporary compatibility shim is acceptable for old imports, but new application code must import the agent-owned implementation.
### Guardrails specification
### Mandatory judge execution for transactions
> Consolidated from `specs/SPEC-005-Guardrails.md`.
> Content consolidated from `docs/JUDGES_TRANSACTIONAL_SAMPLING_FIX.md`.
### Escopo
### Problem
Guardrails são políticas executadas sobre entrada, saída, tool calls, RAG e respostas finais. A plataforma suporta guardrails globais, por agente, por canal e por fase.
Even with `always_run_for_transactional: true`, judges could be skipped by sampling because the `judge` node sent only `context`, `route`, `intent`, and `mcp_results`. Transactional fields produced by the runtime did not reach `JudgePipeline`.
### Fases
### Fix
| Fase | Entrada | Saída |
|---|---|---|
| Input | `user_text`, `context` | `sanitized_input`, `GuardrailResult` |
| Tool | `ToolInvocation` | tool permitida/bloqueada |
| RAG | query/contexto recuperado | contexto aprovado/filtrado |
| Output | `response_text` | resposta aprovada/sanitizada/bloqueada |
| Review | resposta + evidências | decisão final |
The `judge` node now passes:
### GuardrailResult
- `transaction_status`
- `confirmation_required`
- `confirmation_received`
- `tool_policy_result`
- `selected_tool_call`
- `pending_tool_call`
- `mcp_results` as evidence
```json
{
"code": "PINJ",
"phase": "input",
"status": "blocked",
"severity": "high",
"score": 0.98,
"message": "Entrada bloqueada por política.",
"details": {
"matched_policy": "prompt_injection"
}
}
```
`JudgePipeline` detects transactions through multiple signals and evaluates `always_run_for_transactional` before applying `sample_rate`.
### Configuração Global
With the configuration below, common queries continue to be sampled at 25%, but `AWAITING_CONFIRMATION`, `COMPLETED`, `FAILED`, or `CANCELLED` turns always run the judges.
```yaml
input:
- code: MSK
enabled: true
mode: enforce
- code: VLOOP
enabled: true
mode: enforce
- code: PINJ
enabled: true
mode: enforce
output:
- code: REVPREC
enabled: true
mode: enforce
- code: DLEX_OUT
enabled: true
mode: enforce
- code: PINJ
enabled: true
mode: observe
enabled: true
sample_rate: 0.25
always_run_for_transactional: true
```
### Configuração por Agente
### Global Supervisor validation
```yaml
agents:
telecom_contas:
input:
- code: BILLING_INPUT_POLICY
enabled: true
mode: observe
output:
- code: BILLING_COMPLIANCE
enabled: true
mode: enforce
```
> Content consolidated from `docs/docs_GLOBAL_SUPERVISOR_VALIDATION.txt`.
### Modos
VALIDATION - GLOBAL SUPERVISOR
| Modo | Comportamento |
|---|---|
| `enforce` | Aplica bloqueio, máscara ou alteração. |
| `observe` | Registra sem bloquear. |
| `fail_open` | Em erro técnico, prossegue e emite NOC. |
| `fail_closed` | Em erro técnico, bloqueia. |
Implemented changes:
### Tipos
1. Framework
- agent_framework.global_supervisor.models
- agent_framework.global_supervisor.config
- agent_framework.global_supervisor.session_store
- agent_framework.global_supervisor.router
- agent_framework.global_supervisor.client
| Tipo | Implementação |
|---|---|
| Determinístico | Regex, listas, tamanho, estrutura, regras. |
| LLM | Classificação semântica por profile. |
| Híbrido | Determinístico + LLM em casos ambíguos. |
2. New service
- agent_gateway/app/main.py
- agent_gateway/app/settings.py
- agent_gateway/config/backends.yaml
- agent_gateway/README.md
- agent_gateway/Dockerfile
- agent_gateway/docs/ARQUITETURA_GLOBAL_SUPERVISOR.md
### Profiles LLM
3. Docker Compose
- agent-gateway service added on port 8010.
```yaml
profiles:
guardrail:
provider: oci_openai
model: openai.gpt-4.1
temperature: 0
max_tokens: 600
Validations performed:
grl:
provider: oci_openai
model: openai.gpt-4.1
temperature: 0
max_tokens: 700
```
- python3 -m compileall -q agent_framework/src/agent_framework/global_supervisor agent_gateway/app
Result: OK
### Fluxo
- Hybrid-routing smoke test:
Input 1: "My bill is too high" -> billing
Input 2: "and this amount?" on the same session_id -> billing via active_backend
Result: OK
```mermaid
flowchart TD
A[Input] --> B[Deterministic Guardrails]
B --> C{Blocked?}
C -- yes --> D[Safe Response]
C -- no --> E[LLM Guardrails]
E --> F{Approved?}
F -- no --> D
F -- yes --> G[Runtime]
```
- FastAPI app import smoke test:
from app.main import app, registry, router
Result: OK
### Eventos
Note:
- The gateway SSE proxy was left as a future step. The `/gateway/message/sse` endpoint already routes and forwards as a normal message; for end-to-end SSE, a proxy from `/gateway/events/{session_id}` to the active backend can be implemented.
| Evento | Descrição |
|---|---|
| `guardrail.started` | Execução iniciada. |
| `guardrail.completed` | Execução concluída. |
| `guardrail.blocked` | Conteúdo bloqueado. |
| `guardrail.masked` | Conteúdo mascarado. |
| `guardrail.failed` | Falha técnica. |
| `guardrail.observe` | Política observacional registrada. |
### Guardrail event validation
### Códigos Base
> Content consolidated from `docs/docs_VALIDATION_GUARDRAILS_IC.txt`.
| Código | Fase | Uso |
|---|---|---|
| `MSK` | input/output | Mascaramento. |
| `VLOOP` | input | Detecção de loop. |
| `PINJ` | input/output | Prompt injection. |
| `REVPREC` | output | Revisão de precisão. |
| `DLEX_OUT` | output | Controle de dados e linguagem na saída. |
| `RAGSEC` | rag/output | Segurança de contexto recuperado. |
VALIDATION REPORT - guardrails parallel fail-fast + observer IC
Date: 2026-06-03
### Testes
compileall: OK
smoke-tests: OK
| Teste | Objetivo |
|---|---|
| Unitário | Validar guardrail isolado. |
| Config | Validar YAML e schema. |
| Integração | Validar execução no workflow. |
| Observabilidade | Validar eventos e traces. |
| Negativo | Validar bloqueio. |
| Observe-only | Validar não bloqueio. |
### Source files
The files below were consolidated into this manual:
### Requisitos Não Funcionais
- `Documentacao/README_GUARDRAILS_IMPLEMENTADOS.md`
- `docs/EXTERNAL_GUARDRAILS_JUDGES.md`
- `docs/JUDGES_TRANSACTIONAL_SAMPLING_FIX.md`
- `docs/docs_GLOBAL_SUPERVISOR_VALIDATION.txt`
- `docs/docs_VALIDATION_GUARDRAILS_IC.txt`
| Categoria | Requisito |
|---|---|
| Disponibilidade | Componentes deployáveis expõem `/health` e `/ready`. |
| Escalabilidade | Apps stateless escalam horizontalmente. Estado conversacional fica em repositórios externos. |
| Segurança | Segredos são fornecidos por secret store ou Kubernetes Secrets. |
| Observabilidade | Logs, métricas e traces usam correlação por request_id, trace_id, session_id, tenant_id e agent_id. |
| Auditabilidade | Decisões de rota, guardrail, judge, MCP e LLM são rastreáveis. |
| Portabilidade | Execução suportada em local, Docker Compose e Kubernetes/OKE. |
| Configuração | Comportamento variável é controlado por `.env` e YAML versionado. |
### Maintenance rule
### Critérios de Aceite
- [ ] Guardrails globais são carregados por YAML.
- [ ] Guardrails por agente sobrescrevem ou complementam globais.
- [ ] GuardrailResult é gerado para cada execução.
- [ ] Modo enforce bloqueia quando aplicável.
- [ ] Modo observe não bloqueia.
- [ ] Falhas técnicas seguem política configurada.
- [ ] Guardrails LLM usam profile dedicado.
- [ ] Eventos e métricas são emitidos.
- [ ] Testes cobrem casos positivos e negativos.
- [ ] Output guardrails executam antes da resposta final.
### Glossário
| Termo | Definição |
|---|---|
| Agent Platform | Plataforma composta por runtime, gateways, evaluator, templates, contratos e componentes operacionais. |
| Agent Framework | Biblioteca/core reutilizável com contratos, guardrails, judges, memória, telemetria, providers e utilitários. |
| Agent Runtime | Motor de execução de agentes baseado em LangGraph, estado, sessão, memória, checkpoints, roteamento e ciclo de vida. |
| Agent Gateway | Aplicação deployável de entrada, roteamento e orquestração entre backends/agentes. |
| Channel Gateway | Aplicação ou módulo de normalização de payloads de canais para GatewayRequest. |
| AI Gateway | Aplicação de governança, roteamento e abstração de chamadas LLM/embedding. |
| MCP Gateway | Aplicação de governança e roteamento de tools MCP. |
| Evaluator | Camada de avaliação online/offline, regressão e certificação. |
| Business Context | Conjunto de chaves canônicas de negócio: customer_key, contract_key, interaction_key, account_key, resource_key e session_key. |
### Evaluation specification
> Consolidated from `specs/SPEC-006-Evals.md`.
### Escopo
A camada de Evals executa avaliação online, avaliação offline, regressão, certificação e publicação de métricas. Ela padroniza a validação de agentes, prompts, tools, respostas e guardrails.
### Componentes
| Componente | Responsabilidade |
|---|---|
| Online Judges | Avaliação durante a execução. |
| Offline Evaluator | Avaliação batch de conversas. |
| Dataset Runner | Execução de datasets versionados. |
| Regression Runner | Comparação entre versões. |
| Certification Suite | Validação técnica e funcional. |
| Metrics Engine | Cálculo de métricas. |
| Persistence | Persistência de runs e itens. |
| Exporter | Exportação TXT.GZ/JSON/HTML. |
| Publisher | Publicação de scores no Langfuse. |
### Fluxo Offline
```mermaid
flowchart TD
A[Start EvaluationRun] --> B[Collect Conversations]
B --> C[Normalize Items]
C --> D[Run Judges]
D --> E[Calculate Metrics]
E --> F[Persist Results]
F --> G[Export Reports]
G --> H[Publish Scores]
H --> I[Complete Run]
```
### EvaluationRun
```json
{
"run_id": "eval-20260619-001",
"agent_id": "telecom_contas",
"source": "langfuse",
"period_start": "2026-06-18T00:00:00Z",
"period_end": "2026-06-19T00:00:00Z",
"status": "running",
"limit": 500,
"metadata": {
"profile": "judge",
"dataset": "production-sample"
}
}
```
### EvaluationItem
```json
{
"conversation_id": "default:telecom_contas:session-001",
"trace_id": "trace-001",
"agent_id": "telecom_contas",
"input": "Quero consultar minha fatura",
"output": "Sua fatura está aberta...",
"evidence": {
"mcp_results": [],
"rag_context": ""
},
"scores": {
"quality": 0.86,
"groundedness": 0.78,
"safety": 1.0,
"resolution": 0.91
},
"findings": []
}
```
### Métricas
| Métrica | Descrição | Faixa |
|---|---|---|
| `quality` | Clareza, completude e utilidade. | 01 |
| `groundedness` | Aderência a evidências MCP/RAG. | 01 |
| `safety` | Conformidade de segurança. | 01 |
| `resolution` | Capacidade de resolver a intenção. | 01 |
| `tool_correctness` | Uso correto de tools. | 01 |
| `policy_compliance` | Aderência a regras de domínio. | 01 |
### Dataset
```yaml
dataset:
name: telecom_contas_billing
version: 1.0.0
items:
- id: billing-001
input: "Quero consultar minha fatura"
business_context:
customer_key: "11999999999"
contract_key: "3000131180"
expected:
route: billing_agent
tools:
- consultar_fatura
min_scores:
quality: 0.75
groundedness: 0.70
safety: 1.0
```
### Judges
```yaml
judges:
- name: response_quality
enabled: true
threshold: 0.7
profile: judge
- name: groundedness
enabled: true
threshold: 0.6
profile: judge
- name: safety
enabled: true
threshold: 1.0
profile: judge
```
### CLI
```bash
af-evaluator run \
--agent-id telecom_contas \
--source langfuse \
--period-start 2026-06-18T00:00:00Z \
--period-end 2026-06-19T00:00:00Z \
--limit 500
```
### API
| Método | Endpoint | Descrição |
|---|---|---|
| `POST` | `/evaluation/runs` | Cria run. |
| `GET` | `/evaluation/runs/{run_id}` | Consulta run. |
| `GET` | `/evaluation/runs/{run_id}/items` | Lista itens. |
| `POST` | `/evaluation/datasets/{name}/run` | Executa dataset. |
| `GET` | `/health` | Health check. |
### Persistência
| Tabela | Conteúdo |
|---|---|
| `EVAL_RUNS` | Runs executadas. |
| `EVAL_ITEMS` | Conversas avaliadas. |
| `EVAL_SCORES` | Scores por métrica. |
| `EVAL_FINDINGS` | Achados. |
| `EVAL_EXPORTS` | Arquivos exportados. |
### Certificação
A Certification Suite valida:
- endpoints de health;
- GatewayRequest;
- roteamento;
- MCP tools;
- guardrails;
- judges;
- memória;
- checkpoint;
- Langfuse/OTEL;
- datasets mínimos;
- evidências JSON/HTML.
### Eventos
| Evento | Descrição |
|---|---|
| `eval.run.started` | Run iniciada. |
| `eval.item.completed` | Item avaliado. |
| `eval.run.completed` | Run concluída. |
| `eval.run.failed` | Run falhou. |
| `eval.score.published` | Score publicado. |
### Requisitos Não Funcionais
| Categoria | Requisito |
|---|---|
| Disponibilidade | Componentes deployáveis expõem `/health` e `/ready`. |
| Escalabilidade | Apps stateless escalam horizontalmente. Estado conversacional fica em repositórios externos. |
| Segurança | Segredos são fornecidos por secret store ou Kubernetes Secrets. |
| Observabilidade | Logs, métricas e traces usam correlação por request_id, trace_id, session_id, tenant_id e agent_id. |
| Auditabilidade | Decisões de rota, guardrail, judge, MCP e LLM são rastreáveis. |
| Portabilidade | Execução suportada em local, Docker Compose e Kubernetes/OKE. |
| Configuração | Comportamento variável é controlado por `.env` e YAML versionado. |
### Critérios de Aceite
- [ ] Evaluator executa runs por período/agente.
- [ ] Langfuse é fonte suportada.
- [ ] Datasets são versionados.
- [ ] LLM Judges usam profile `judge`.
- [ ] Scores são persistidos.
- [ ] TXT.GZ/JSON/HTML são exportáveis.
- [ ] Scores podem ser publicados no Langfuse.
- [ ] Certification Suite gera evidências.
- [ ] Métricas mínimas são padronizadas.
- [ ] Falhas permitem retomada por checkpoint de run.
### Glossário
| Termo | Definição |
|---|---|
| Agent Platform | Plataforma composta por runtime, gateways, evaluator, templates, contratos e componentes operacionais. |
| Agent Framework | Biblioteca/core reutilizável com contratos, guardrails, judges, memória, telemetria, providers e utilitários. |
| Agent Runtime | Motor de execução de agentes baseado em LangGraph, estado, sessão, memória, checkpoints, roteamento e ciclo de vida. |
| Agent Gateway | Aplicação deployável de entrada, roteamento e orquestração entre backends/agentes. |
| Channel Gateway | Aplicação ou módulo de normalização de payloads de canais para GatewayRequest. |
| AI Gateway | Aplicação de governança, roteamento e abstração de chamadas LLM/embedding. |
| MCP Gateway | Aplicação de governança e roteamento de tools MCP. |
| Evaluator | Camada de avaliação online/offline, regressão e certificação. |
| Business Context | Conjunto de chaves canônicas de negócio: customer_key, contract_key, interaction_key, account_key, resource_key e session_key. |
### Evaluation and certification framework
> Consolidated from `specs/SPEC-019-Evaluation-and-Certification-Framework.md`.
### Agent Platform OCI
Version: 1.0.0
---
### Padrão de leitura
Cada SPEC está organizada para servir tanto como contrato arquitetural quanto como guia prático de adoção.
A estrutura usada é:
1. Conceito.
2. Problema que resolve.
3. Quando usar.
4. Quando não usar.
5. Arquitetura.
6. Implementação.
7. Exemplos.
8. Erros comuns.
9. Critérios de aceite.
---
### 1. Conceito
Evaluation mede qualidade e comportamento. Certification valida prontidão técnica e funcional.
Evaluator responde:
```text
O agente respondeu bem?
A resposta está fundamentada?
A tool certa foi chamada?
Houve regressão?
```
Certification responde:
```text
O agente está pronto para rodar?
Endpoints funcionam?
MCP funciona?
Guardrails funcionam?
Observabilidade funciona?
```
### 2. Arquitetura
```mermaid
flowchart LR
Runtime[Runtime] --> LF[Langfuse]
LF --> Eval[Offline Evaluator]
Dataset[Datasets] --> Eval
Eval --> Scores[Scores]
Eval --> Reports[Reports]
Cert[Certification Suite] --> Runtime
Cert --> Evidence[Evidences]
```
### 3. Métricas
| Métrica | Descrição |
| --- | --- |
| quality | Clareza, completude e utilidade. |
| groundedness | Aderência a evidências MCP/RAG. |
| safety | Conformidade de segurança. |
| resolution | Resolve a intenção. |
| tool_correctness | Usa tools corretas. |
| route_accuracy | Rota/intenção corretas. |
| policy_compliance | Aderência à política de domínio. |
### 4. Dataset
```yaml
dataset:
name: telecom_contas_regression
version: 1.0.0
items:
- id: billing-001
input: "Quero consultar minha fatura"
business_context:
customer_key: "11999999999"
contract_key: "3000131180"
expected:
route: billing_agent
tools:
- consultar_fatura
min_scores:
quality: 0.75
groundedness: 0.70
```
### 5. EvaluationRun
```json
{
"run_id": "eval-001",
"agent_id": "telecom_contas",
"source": "langfuse",
"period_start": "2026-06-18T00:00:00Z",
"period_end": "2026-06-19T00:00:00Z",
"status": "running"
}
```
### 6. CLI
```bash
af-evaluator run --agent-id telecom_contas --dataset datasets/telecom_contas.yaml
```
### 7. Certification
Valida:
- health;
- GatewayRequest;
- routing;
- identity;
- MCP;
- RAG;
- guardrails;
- judges;
- memory;
- checkpoint;
- Langfuse;
- OTEL.
### 8. Evidências
- JSON;
- HTML;
- TXT.GZ legado;
- scores Langfuse;
- logs;
- traces;
- screenshots quando aplicável.
### 9. Erros comuns
| Erro | Impacto | Correção |
| --- | --- | --- |
| Dataset só com casos felizes | Baixa cobertura. | Incluir negativos e bordas. |
| Evaluator sem baseline | Sem comparação. | Registrar baseline. |
| Certification sem MCP real/mock | Integração não validada. | Criar tool test. |
| Judge sem threshold | Sem critério objetivo. | Definir threshold. |
### 10. Critérios de aceite
- [ ] Dataset versionado.
- [ ] Evaluator executado.
- [ ] Scores persistidos.
- [ ] Certification executada.
- [ ] Relatórios gerados.
- [ ] Thresholds definidos.
- [ ] Casos negativos incluídos.
- [ ] Scores publicados quando aplicável.
New fixes or evolutions for this subject should update this consolidated document. Release notes may continue to exist as history, but they should not be required to understand or implement the feature.

View File

@@ -1,73 +1,41 @@
### RAG, BusinessContext and Grounding
### RAG, BusinessContext, and Grounding
### How to use this manual
This is a **specialized reference manual**. It does not replace the main tutorial.
- To build an agent end to end, use [`README_en.md`](../../../README_en.md).
- Use this document when implementing, deep-diving or troubleshooting **RAG, providers, BusinessContext, retrieved context and grounding**.
- Historical examples consolidated here must be interpreted against the current framework API.
- If documentation differs, the current code and root README take precedence.
- To create an agent from start to finish, use [`README_en.md`](../../../README_en.md).
- Use this document when you need to implement, deepen, or diagnose **RAG, providers, BusinessContext, retrieved context, and grounding**.
- Historical examples consolidated here should be read in light of the framework's current API.
- In case of divergence, the code for the version and the current `README_en.md` take precedence.
### Relationship with the main tutorial
`README_en.md` introduces this capability as part of the normal development flow. This manual consolidates details previously spread across `docs/`, `Documentacao/`, release notes, validation records and specialized guides.
The `README_en.md` presents this capability in the normal development flow. This manual brings together details that were distributed across `docs/`, `Documentacao/`, release notes, validations, and specialized guides.
Its purpose is to answer **“how does this feature work in depth and how do I troubleshoot it?”** without becoming a second copy of the main tutorial.
The goal here is to answer **“how does this feature work in depth and how do I solve problems with it?”**, without turning this file into a second copy of the main tutorial.
### Scope
Rag, providers, businesscontext, retrieved context and grounding.
RAG, providers, BusinessContext, retrieved context, and grounding.
### Consolidated technical content
### RAG, Enterprise Providers, BusinessContext and Grounding
### RAG, Enterprise Providers, BusinessContext, and Grounding
This guide covers configurable retrieval and its relationship with tools, memory and agent context.
Guide for integrating retrieved knowledge, selecting between RAG providers, configuring KBDB, using samples, MCP sufficiency, and using BusinessContext as a data contract.
### Provider selection
### How to use this document
RAG is provider-based. The standard implementation and the enterprise KBDB implementation are selected through configuration rather than through domain branches in the agent code. Provider-specific connection/index settings remain environment/configuration concerns.
This is the consolidated development document for this subject. It brings together architecture, configuration, examples, runtime behavior, compatibility, tests, and troubleshooting that were previously distributed across several files. Source sections were preserved when they provided distinct technical details; release notes were incorporated as current behavior or correction history.
### Runtime role
### Standard RAG Provider versus KBDB Enterprise
Retrieved knowledge is injected into the execution context so the agent can ground informational responses. RAG does not replace transactional tool execution and it is not the same as long-term memory. Use RAG for external/reference knowledge, MCP for live business operations/data and LTM for durable user/customer facts.
> Content consolidated from `docs/RAG_PROVIDER_KBDB.md`.
### KBDB Enterprise
The framework now supports two retrieval backends through the same `RagService` contract, without changing agents or `_retrieve_rag_context()`.
The KBDB provider is an alternative backend with its own configuration while preserving the framework-facing retrieval contract. Agent code should not need to know which provider is active.
### BusinessContext
BusinessContext v2 carries generic business identifiers resolved from domain aliases. RAG filters, tool calls and telemetry can consume these canonical keys without introducing `msisdn`, invoice/order naming or other domain fields into shared modules.
### MCP sufficiency and grounding
When a tool result already contains sufficient authoritative data for the requested answer, the runtime can avoid unnecessary retrieval/composition work according to the configured response path. Conversely, a RAG answer must not claim a transactional action occurred merely because documentation describes how the action works.
### Sample validation
The project contains sample PDFs/policies for billing, orders, products, support and business-context/RAG flow. Use them to validate ingestion/embedding/retrieval and ask targeted questions whose expected answer is present in one document.
### Source material consolidated
- `docs/RAG_PROVIDER_KBDB.md`
- `docs/README_rag_samples.md`
- `Documentacao/README_TEMPLATE_BUSINESS_CONTEXT_V2.md`
- operational RAG/cache notes in `Documentacao/README_FIRST_MAX_OPERATIONAL_FIXES.md`
### Detailed normative and implementation reference
The sections below preserve the detailed English project specifications and implementation guides relevant to this capability. They are included here so a developer does not need to reconstruct the behavior from separate documents.
### RAG provider implementation notes
> Consolidated from `docs/RAG_PROVIDER_KBDB.md`.
O framework passa a suportar dois backends de retrieval pelo mesmo contrato `RagService`, sem alterar os agentes nem `_retrieve_rag_context()`.
### Seleção
### Selection
```env
RAG_PROVIDER=standard # default: comportamento anterior
@@ -75,23 +43,23 @@ RAG_PROVIDER=standard # default: comportamento anterior
RAG_PROVIDER=kbdb # KBDB enterprise
```
A seleção é exclusiva por processo. Os dois RAGs não executam juntos e não compartilham vector store, graph store ou ingestão.
Selection is exclusive per process. The two RAG implementations do not run together and do not share vector store, graph store, or ingestion.
### `standard`
Mantém integralmente o RAG já existente no `agent_framework_oci`: `VECTOR_STORE_PROVIDER`, `GRAPH_STORE_PROVIDER`, embedding, query rewrite, compression, retrieval guardrails e geração continuam válidos.
Fully preserves the existing RAG in `agent_framework_oci`: `VECTOR_STORE_PROVIDER`, `GRAPH_STORE_PROVIDER`, embedding, query rewrite, compression, retrieval guardrails, and generation remain valid.
### `kbdb`
O framework integra somente a porta estável de serving do projeto KBDB:
The framework integrates only the stable serving port of the KBDB project:
`PKG_KB_SERVING.SEARCH_KNOWLEDGE_BASE`
O pipeline enterprise continua externo ao runtime do agente e preserva sua própria arquitetura RAW → SILVER → GOLD, HVI/hybrid search, property graph, publicação, lifecycle, auditoria e observabilidade.
The enterprise pipeline remains external to the agent runtime and preserves its own RAW → SILVER → GOLD architecture, HVI/hybrid search, property graph, publishing, lifecycle, audit, and observability.
O envelope KBDB é adaptado para `RagResult`/`VectorDocument`; portanto os agentes existentes continuam chamando `_retrieve_rag_context()` e os retrieval guardrails do framework continuam depois do retrieval.
The KBDB envelope is adapted to `RagResult`/`VectorDocument`; therefore existing agents continue calling `_retrieve_rag_context()` and the framework's retrieval guardrails continue after retrieval.
### Configuração
### Configuration
```env
RAG_PROVIDER=kbdb
@@ -111,23 +79,23 @@ KBDB_METADATA_JSON=
KBDB_MIN_SCORE=
```
Quando `RAG_PROVIDER=kbdb`, `KBDB_DB_USER`, `KBDB_DB_PASSWORD` e `KBDB_DB_DSN` são obrigatórios. O KBDB usa conexão isolada porque pode residir em outro Autonomous. `KBDB_DB_DSN` segue a mesma semântica de `ADB_DSN`: use o alias TNS existente no `tnsnames.ora` da wallet indicada por `KBDB_DB_WALLET_LOCATION`, e não uma URL `tcps://...`.
When `RAG_PROVIDER=kbdb`, `KBDB_DB_USER`, `KBDB_DB_PASSWORD`, and `KBDB_DB_DSN` are required. KBDB uses an isolated connection because it may reside in another Autonomous database. `KBDB_DB_DSN` follows the same semantics as `ADB_DSN`: use the existing TNS alias in the `tnsnames.ora` from the wallet indicated by `KBDB_DB_WALLET_LOCATION`, not a `tcps://...` URL.
### Isolamento e compatibilidade
### Isolation and compatibility
- `RAG_PROVIDER=standard` não importa nem conecta ao KBDB.
- `RAG_PROVIDER=kbdb` não instancia vector/graph stores do RAG padrão.
- Ingestão por `RagService.add_documents()` não é permitida no modo KBDB: deve passar pelo pipeline/publicação KBDB.
- Query rewrite e context compression continuam opcionais e são aplicados pela camada comum do framework.
- `AgentRuntimeMixin._retrieve_rag_context()` e os agentes permanecem inalterados.
- Falhas do KBDB seguem a semântica existente do framework: retrieval é evidência auxiliar e a exceção é convertida em metadata técnica sem derrubar a jornada.
- `RAG_PROVIDER=standard` does not import or connect to KBDB.
- `RAG_PROVIDER=kbdb` does not instantiate the standard RAG vector/graph stores.
- Ingestion through `RagService.add_documents()` is not allowed in KBDB mode: it must go through the KBDB pipeline/publishing process.
- Query rewrite and context compression remain optional and are applied by the framework's common layer.
- `AgentRuntimeMixin._retrieve_rag_context()` and agents remain unchanged.
- KBDB failures follow the framework's existing semantics: retrieval is auxiliary evidence and the exception is converted into technical metadata without breaking the user journey.
### Resposta direta de tool e RAG
### Direct tool response and RAG
O framework não considera mais que um resultado MCP estruturado é, por si só, uma resposta suficiente ao usuário.
The framework no longer considers a structured MCP result, by itself, to be a sufficient user response.
Uma política `response.renderer` define somente **como** apresentar o resultado. Ela não encerra o fluxo antes de RAG/LLM. Para uma tool deliberadamente produzir uma resposta final direta, a aplicação deve declarar explicitamente:
A `response.renderer` policy defines only **how** to present the result. It does not terminate the flow before RAG/LLM. For a tool to deliberately produce a direct final response, the application must explicitly declare:
```yaml
response:
@@ -136,31 +104,23 @@ response:
direct: true
```
Sem `direct: true`, o resultado da tool permanece como evidência MCP e o fluxo segue para `_retrieve_rag_context()` e composição LLM. Isso permite, por exemplo, que uma consulta operacional de plano seja combinada com conhecimento documental do KBDB quando a pergunta pedir regras, políticas ou explicações.
Without `direct: true`, the tool result remains MCP evidence and the flow continues to `_retrieve_rag_context()` and LLM composition. This allows, for example, an operational plan query to be combined with KBDB documentary knowledge when the question asks for rules, policies, or explanations.
O core do framework não possui fallback por nome de tool (`consultar_plano`, `consultar_pedido`, etc.). Regras de apresentação pertencem à aplicação/domínio.
The framework core has no fallback by tool name (`consultar_plano`, `consultar_pedido`, etc.). Presentation rules belong to the application/domain.
### Suficiência MCP e grounding
### MCP sufficiency and grounding
Um resultado MCP bem-sucedido **não** faz o framework pular RAG automaticamente.
O domínio só pode declarar suficiência documental explicitamente no payload com
`rag_sufficient=true` ou `knowledge_sufficient=true`. Essa decisão é genérica e
não depende do nome da tool nem de palavras-chave de telecom/retail.
A successful MCP result **does not** make the framework skip RAG automatically.
The domain may declare documentary sufficiency only explicitly in the payload with `rag_sufficient=true` or `knowledge_sufficient=true`. This decision is generic and does not depend on the tool name or telecom/retail keywords.
No provider `kbdb`, `KBDB_GROUNDED_ONLY=true` é o padrão. Quando a busca KBDB
retorna vazia, bloqueada ou com erro, a composição LLM pode usar fatos comprovados
por MCP/business context, mas não pode completar a parte documental com conhecimento
paramétrico do modelo. Deve informar que não há evidência suficiente na base.
For the `kbdb` provider, `KBDB_GROUNDED_ONLY=true` is the default. When KBDB search returns empty, blocked, or error, LLM composition may use facts proven by MCP/business context, but it must not fill the documentary portion using parametric model knowledge. It must state that there is insufficient evidence in the knowledge base.
Eventos do ProductAgent registram `IC.PRODUCT_RAG_CONTEXT_EVALUATED` em toda
tentativa/decisão e `IC.PRODUCT_RAG_CONTEXT_RETRIEVED` somente quando há contexto
recuperado. Os metadados incluem `provider`, `status`, `document_count`, `reason`,
`error`, `query`, `namespace` e `latency_ms`.
ProductAgent events record `IC.PRODUCT_RAG_CONTEXT_EVALUATED` for every attempt/decision and `IC.PRODUCT_RAG_CONTEXT_RETRIEVED` only when context was retrieved. Metadata includes `provider`, `status`, `document_count`, `reason`, `error`, `query`, `namespace`, and `latency_ms`.
### RAG sample validation guide
### RAG samples and tests
> Consolidated from `docs/README_rag_samples.md`.
> Content consolidated from `docs/README_rag_samples.md`.
These PDF files are synthetic, searchable sample documents created to validate the RAG embedding and retrieval flow of `agent_template_backend`.
@@ -216,588 +176,212 @@ OCI_EMBEDDING_MODEL=cohere.embed-multilingual-v3.0
- What is the target response for a critical support ticket?
- How does BusinessContext map customer_key to MCP tool parameters?
### Runtime integration constraints
### BusinessContext v2
> Consolidated from `specs/SPEC-002-Agent-Runtime.md`.
> Content consolidated from `Documentacao/README_TEMPLATE_BUSINESS_CONTEXT_V2.md`.
### Escopo
This package updates `agent_template_backend` and `agent_frontend` to reflect the new framework, where keys coming from the channel/front end are resolved once into canonical keys and propagated through the layers to the MCP Server.
O Agent Runtime executa o ciclo de vida conversacional do agente. A execução inclui normalização de contexto, estado LangGraph, memória, checkpoint, roteamento, supervisor, guardrails, MCP, RAG, LLM, judges, persistência e resposta final.
### Implemented flow
### Componentes
1. The front end sends `tenant_id`, `agent_id`, `session_id`, and `business_context`.
2. The backend normalizes the message through `ChannelGateway`, preserving the full payload in `context`.
3. The backend uses `IdentityResolver` with `config/identity.yaml` to generate `BusinessContext`:
- `customer_key`
- `contract_key`
- `interaction_key`
- `account_key`
- `resource_key`
- `session_key`
4. The workflow receives `context.business_context`.
5. Example agents no longer build specific arguments such as `msisdn`, `invoice_id`, or `order_id` directly.
6. `MCPToolRouter` uses `config/mcp_parameter_mapping.yaml` to convert canonical keys into the actual parameters of each MCP tool.
| Componente | Responsabilidade |
|---|---|
| Workflow Builder | Compila o grafo LangGraph. |
| State Manager | Mantém o estado de execução. |
| Session Manager | Resolve sessão e conversation_key. |
| Memory Manager | Carrega e persiste histórico. |
| Checkpoint Manager | Persiste estado LangGraph. |
| Input Guardrail Node | Executa guardrails de entrada. |
| Router Node | Decide rota/intent. |
| Supervisor Node | Decide handoff ou próximo agente quando habilitado. |
| Agent Node | Executa agente de domínio. |
| MCP Client/Router | Executa tools por contrato. |
| RAG Service | Recupera contexto documental. |
| Output Supervisor | Revisa resposta antes de saída. |
| Output Guardrail Node | Executa guardrails de saída. |
| Judge Node | Avalia resposta. |
| Persistence Node | Persiste mensagens, memória e checkpoint. |
### Main files adjusted
### State Model
- `agent_template_backend/app/main.py`
- loads `IdentityResolver`;
- resolves `BusinessContext` per message;
- persists keys in session/memory/metadata/SSE;
- adds `/debug/identity`.
```python
class AgentState(TypedDict, total=False):
user_text: str
sanitized_input: str
response_text: str
tenant_id: str
agent_id: str
channel: str
session_id: str
conversation_key: str
message_id: str
route: str
intent: str
context: dict
business_context: dict
tool_arguments: dict
mcp_tools: list[str]
mcp_results: list[dict]
rag_context: str
rag_metadata: dict
guardrails: list[dict]
judges: list[dict]
metadata: dict
errors: list[dict]
- `agent_template_backend/app/agents/runtime.py`
- adds centralized `_collect_mcp_context()`;
- forwards `business_context` and `original_context` to the MCP Router.
- `agent_template_backend/app/agents/*_agent.py`
- agents now use `_collect_mcp_context()` instead of building specific arguments.
- `agent_template_backend/config/identity.yaml`
- defines how channel/front-end fields feed canonical keys.
- `agent_template_backend/config/mcp_parameter_mapping.yaml`
- defines how canonical keys become real parameters per MCP tool.
- `agent_frontend/index.html` and `agent_frontend/app.js`
- add `tenant`, `agent`, and canonical-key fields;
- send `business_context` in the payload;
- retain domain aliases for compatibility (`msisdn`, `invoice_id`, `order_id`, etc.).
### Quick test
Start backend, frontend, and MCP servers. Then test:
```bash
curl -s http://localhost:8000/health | jq
curl -s -X POST http://localhost:8000/debug/identity \
-H 'Content-Type: application/json' \
-d '{
"channel":"web",
"tenant_id":"default",
"agent_id":"telecom_contas",
"payload":{
"message":"Minha fatura veio alta",
"session_id":"teste-001",
"msisdn":"11999999999",
"invoice_id":"3000131180",
"ura_call_id":"URA-123",
"business_context":{
"customer_key":"11999999999",
"contract_key":"3000131180",
"interaction_key":"URA-123",
"session_key":"teste-001"
}
}
}' | jq
curl -s -X POST http://localhost:8000/debug/mcp/call/consultar_fatura \
-H 'Content-Type: application/json' \
-d '{
"business_context": {
"customer_key":"11999999999",
"contract_key":"3000131180",
"interaction_key":"URA-123",
"session_key":"teste-001"
}
}' | jq
```
### Workflow
In the backend log, look for `mcp.tool.mapped`. It should indicate the mapped keys and `has_msisdn=true`, `has_invoice_id=true` for the telecom domain.
```mermaid
flowchart TD
A[start] --> B[input_guardrails]
B --> C[routing_decision]
C --> D[agent_execution]
D --> E[output_supervisor]
E --> F[output_guardrails]
F --> G[judge]
G --> H[persist]
H --> I[end]
C --> J[handoff]
J --> C
```
### Operational RAG and cache integration
### Nós
> Content consolidated from `Documentacao/README_FIRST_MAX_OPERATIONAL_FIXES.md`.
| Nó | Entrada | Saída |
|---|---|---|
| `input_guardrails` | `user_text`, `context` | `sanitized_input`, `guardrails` |
| `routing_decision` | `sanitized_input`, `business_context` | `route`, `intent`, `mcp_tools` |
| `agent_execution` | `state` completo | `response_text`, `mcp_results`, `rag_metadata` |
| `output_supervisor` | `response_text` | `response_text` revisado |
| `output_guardrails` | `response_text` | `response_text`, `guardrails` |
| `judge` | `response_text`, evidências | `judges` |
| `persist` | `state` completo | checkpoint, memória, mensagens |
This version fixes the gaps identified in the comparison against FIRST.
### Router
### Applied fixes
```yaml
routing:
mode: router
fallback_agent: billing_agent
enable_llm_router: false
intents:
billing_invoice_explanation:
route: billing_agent
keywords:
- fatura
- cobrança
- boleto
mcp_tools:
- consultar_fatura
- consultar_pagamentos
```
### 1. Operational LangGraph checkpoint
### Supervisor
```yaml
supervisor:
enabled: true
profile: supervisor
max_turns: 5
handoff_enabled: true
fallback_route: support_agent
```
### Memory
| Provider | Uso |
|---|---|
| `memory` | Execução local e testes. |
| `sqlite` | Desenvolvimento local persistente. |
| `mongodb` | Checkpoint e histórico em ambiente distribuído. |
| `autonomous` | Produção com Oracle Autonomous Database. |
### Checkpoints
Checkpoint contém:
```json
{
"conversation_key": "default:telecom_contas:session-001",
"checkpoint_id": "ckpt-001",
"state": {},
"pending_writes": [],
"created_at": "2026-06-19T12:00:00Z"
}
```
Formato entregue ao LangGraph:
```python
pending_writes: list[tuple[str, str, object]]
```
### Business Context
```yaml
business_context:
customer_key: "11999999999"
contract_key: "3000131180"
interaction_key: "301953872"
account_key: null
resource_key: null
session_key: "session-001"
metadata:
source_channel: web
```
### Ordem de Prioridade dos Dados
1. `tool_arguments`
2. `business_context`
3. `context`
4. `session.metadata`
5. `state`
6. extração complementar do texto
### MCP Integration
```mermaid
flowchart LR
AgentNode --> ToolList[mcp_tools]
ToolList --> Mapping[mcp_parameter_mapping.yaml]
Mapping --> MCP[MCP Gateway/Router]
MCP --> Result[mcp_results]
```
### RAG Integration
```yaml
rag:
enabled: true
namespace_strategy: agent_id
top_k: 5
profile_generation: rag_generation
```
### Eventos
| Evento | Descrição |
|---|---|
| `runtime.started` | Execução iniciada. |
| `runtime.session.loaded` | Sessão carregada. |
| `runtime.memory.loaded` | Memória carregada. |
| `runtime.checkpoint.loaded` | Checkpoint carregado. |
| `runtime.route.selected` | Rota selecionada. |
| `runtime.agent.started` | Agente iniciado. |
| `runtime.agent.completed` | Agente concluído. |
| `runtime.persist.completed` | Persistência concluída. |
| `runtime.failed` | Falha controlada. |
### Erros
| Código | Condição | Tratamento |
|---|---|---|
| `RUNTIME_INVALID_REQUEST` | GatewayRequest inválido | 422 |
| `RUNTIME_ROUTE_NOT_FOUND` | Nenhuma rota elegível | fallback ou resposta controlada |
| `RUNTIME_CHECKPOINT_ERROR` | Falha em checkpoint | retry ou stateless conforme config |
| `RUNTIME_MEMORY_ERROR` | Falha em memória | retry ou resposta controlada |
| `RUNTIME_AGENT_ERROR` | Falha no agente | NOC + fallback |
| `RUNTIME_TIMEOUT` | Timeout geral | resposta controlada |
### Contrato Durável de Estado Transacional
Hosts que utilizam `AgentRuntime` com transações multi-turno DEVEM declarar no `AgentState` os campos `active_transaction` e `last_transaction`. O primeiro é a fonte canônica da transação em andamento e deve sobreviver a checkpoint/resume; o segundo mantém o snapshot da última transação terminal.
```python
active_transaction: dict[str, Any]
last_transaction: dict[str, Any]
```
`selected_tool_call` e `pending_tool_call` são campos auxiliares/compatibilidade e não substituem o latch canônico. Durante `COLLECTING_PARAMETERS`, a retomada da transação e o consumo de parâmetros pendentes têm precedência sobre keyword routing genérico. Uma mudança de intenção só deve interromper a transação quando for inequívoca ou explicitamente solicitada pelo usuário.
O contrato completo, ciclo de vida, precedência de roteamento, checklist e testes regressivos estão em [`docs/TRANSACTION_STATE_DEVELOPER_GUIDE.md`](../docs/TRANSACTION_STATE_DEVELOPER_GUIDE.md).
### Requisitos Não Funcionais
| Categoria | Requisito |
|---|---|
| Disponibilidade | Componentes deployáveis expõem `/health` e `/ready`. |
| Escalabilidade | Apps stateless escalam horizontalmente. Estado conversacional fica em repositórios externos. |
| Segurança | Segredos são fornecidos por secret store ou Kubernetes Secrets. |
| Observabilidade | Logs, métricas e traces usam correlação por request_id, trace_id, session_id, tenant_id e agent_id. |
| Auditabilidade | Decisões de rota, guardrail, judge, MCP e LLM são rastreáveis. |
| Portabilidade | Execução suportada em local, Docker Compose e Kubernetes/OKE. |
| Configuração | Comportamento variável é controlado por `.env` e YAML versionado. |
### Critérios de Aceite
- [ ] Runtime recebe GatewayRequest validado.
- [ ] State contém tenant_id, agent_id, session_id, conversation_key, route e intent.
- [ ] Input guardrails executam antes do roteamento.
- [ ] Router ou Supervisor seleciona rota.
- [ ] Agent Node executa sem acessar payload bruto de canal.
- [ ] MCP é acessado por contrato.
- [ ] RAG é acessado por serviço reutilizável.
- [ ] Output guardrails executam antes da resposta final.
- [ ] Judges geram JudgeResult.
- [ ] Memória e checkpoint são persistidos conforme provider.
- [ ] Hosts transacionais declaram `active_transaction` e `last_transaction` no `AgentState`.
- [ ] Durante `COLLECTING_PARAMETERS`, respostas a parâmetros pendentes têm precedência sobre keyword routing genérico.
- [ ] Erros geram NOC e resposta controlada.
### Glossário
| Termo | Definição |
|---|---|
| Agent Platform | Plataforma composta por runtime, gateways, evaluator, templates, contratos e componentes operacionais. |
| Agent Framework | Biblioteca/core reutilizável com contratos, guardrails, judges, memória, telemetria, providers e utilitários. |
| Agent Runtime | Motor de execução de agentes baseado em LangGraph, estado, sessão, memória, checkpoints, roteamento e ciclo de vida. |
| Agent Gateway | Aplicação deployável de entrada, roteamento e orquestração entre backends/agentes. |
| Channel Gateway | Aplicação ou módulo de normalização de payloads de canais para GatewayRequest. |
| AI Gateway | Aplicação de governança, roteamento e abstração de chamadas LLM/embedding. |
| MCP Gateway | Aplicação de governança e roteamento de tools MCP. |
| Evaluator | Camada de avaliação online/offline, regressão e certificação. |
| Business Context | Conjunto de chaves canônicas de negócio: customer_key, contract_key, interaction_key, account_key, resource_key e session_key. |
### Business context contracts
> Consolidated from `specs/SPEC-012-Canonical-Contracts.md`.
### Agent Platform OCI
Version: 1.0.0
---
### Padrão de leitura
Cada SPEC está organizada para servir tanto como contrato arquitetural quanto como guia prático de adoção.
A estrutura usada é:
1. Conceito.
2. Problema que resolve.
3. Quando usar.
4. Quando não usar.
5. Arquitetura.
6. Implementação.
7. Exemplos.
8. Erros comuns.
9. Critérios de aceite.
---
### 1. Conceito
Contratos canônicos são estruturas padronizadas usadas para desacoplar canais, gateways, runtime, agentes, tools, LLMs, evaluator e observabilidade.
A plataforma usa contratos para garantir que componentes independentes possam evoluir sem quebrar uns aos outros.
### 2. Problema que resolve
Sem contratos:
- cada canal envia payload diferente;
- agentes passam a conhecer WhatsApp, Voice, Teams ou CRM;
- MCP tools recebem parâmetros inconsistentes;
- LLM calls ficam acopladas ao provider;
- evaluator não consegue comparar respostas;
- observabilidade fica fragmentada.
Com contratos:
The workflow no longer compiles directly with `MemorySaver()`. The following adapter was created:
```text
Canal → GatewayRequest → Runtime → BusinessContext → ToolInvocation → ToolResult
agent_framework/checkpoints/langgraph_saver.py
```
### 3. Catálogo de contratos
It connects LangGraph to the framework's configured repository:
| Contrato | Uso |
| --- | --- |
| GatewayRequest | Entrada canônica da plataforma. |
| ChannelResponse | Resposta canônica ao canal. |
| BusinessContext | Identidade canônica de negócio. |
| AgentState | Estado interno do runtime. |
| Session | Sessão técnica/conversacional. |
| Checkpoint | Persistência de estado LangGraph. |
| ToolInvocation | Chamada canônica de tool MCP. |
| ToolResult | Resposta canônica de tool MCP. |
| LLMRequest | Chamada canônica ao AI Gateway. |
| LLMResponse | Resposta canônica do AI Gateway. |
| EvaluationRun | Execução do evaluator. |
| EvaluationResult | Resultado de avaliação. |
| CertificationResult | Resultado de certificação. |
| EventEnvelope | Envelope de eventos IC/NOC/GRL. |
- `memory`
- `sqlite`
- `oracle` / `autonomous`
### 4. GatewayRequest
### 4.1. Uso
Usado por Channel Gateway e Agent Gateway para enviar mensagens ao Runtime.
```json
{
"channel": "web",
"tenant_id": "default",
"agent_id": "telecom_contas",
"payload": {
"message": "Quero consultar minha fatura",
"session_id": "session-001",
"user_id": "user-001",
"message_id": "msg-001",
"business_context": {
"customer_key": "11999999999",
"contract_key": "3000131180",
"interaction_key": "301953872",
"session_key": "session-001"
},
"metadata": {
"request_id": "req-001",
"contract_version": "gateway-request-v1"
}
}
}
```
### 4.2. Campos obrigatórios
- `channel`;
- `payload.message`;
- `payload.session_id`;
- `payload.message_id`;
- `tenant_id` quando multi-tenant;
- `agent_id` quando não houver roteamento global.
### 5. ChannelResponse
```json
{
"channel": "web",
"session_id": "default:telecom_contas:session-001",
"text": "Resposta final do agente.",
"metadata": {
"tenant_id": "default",
"agent_id": "telecom_contas",
"route": "billing_agent",
"intent": "billing_invoice_explanation",
"guardrails": [],
"judges": []
}
}
```
### 6. BusinessContext
### 6.1. Uso
BusinessContext transporta identidade de negócio sem acoplar a plataforma ao formato de cada canal.
```yaml
business_context:
customer_key: "11999999999"
contract_key: "3000131180"
interaction_key: "301953872"
account_key: null
resource_key: null
session_key: "session-001"
metadata:
source_channel: web
```
### 6.2. Mapeamento para MCP
```yaml
tools:
consultar_fatura:
map:
customer_key: msisdn
contract_key: invoice_id
interaction_key: ura_call_id
session_key: session_id
```
### 7. AgentState
In the workflow:
```python
class AgentState(TypedDict, total=False):
user_text: str
sanitized_input: str
response_text: str
tenant_id: str
agent_id: str
channel: str
session_id: str
conversation_key: str
message_id: str
route: str
intent: str
business_context: dict
mcp_tools: list[str]
mcp_results: list[dict]
rag_context: str
guardrails: list[dict]
judges: list[dict]
builder.compile(checkpointer=create_langgraph_checkpointer(self.settings))
```
### 8. ToolInvocation
### 2. LangGraph telemetry wrapping actual execution
```json
{
"tenant_id": "default",
"agent_id": "telecom_contas",
"tool_name": "consultar_fatura",
"arguments": {
"msisdn": "11999999999",
"invoice_id": "3000131180"
},
"business_context": {
"customer_key": "11999999999",
"contract_key": "3000131180"
},
"metadata": {
"request_id": "req-001",
"trace_id": "trace-001"
}
}
A node wrapper was added to the workflow:
```python
self._node("billing_agent", self.billing_agent)
```
### 9. ToolResult
This way the `langgraph.node.*` span/event wraps actual node execution, not just an empty block.
```json
{
"tool_name": "consultar_fatura",
"ok": true,
"data": {
"invoice_id": "3000131180",
"valor_total": 249.90,
"status": "ABERTA"
},
"cache": {
"hit": false,
"ttl_seconds": 300
},
"latency_ms": 140
}
Events emitted:
- `langgraph.node.started`
- `langgraph.node.completed`
- `langgraph.node.failed`
- `langgraph.edge.selected`
### 3. RAG integrated into agents
Agents now receive `RagService` and use retrieved context in the prompt:
- BillingAgent
- ProductAgent
- OrdersAgent
- SupportAgent
RAG uses:
- `VECTOR_STORE_PROVIDER=memory|sqlite|oracle|autonomous`
- `GRAPH_STORE_PROVIDER=memory|oracle|autonomous`
- `RAG_TOP_K`
### 4. Cache integrated into agent runtime
The following mixin was created:
```text
agent_template_backend/app/agents/runtime.py
```
### 10. LLMRequest
It adds:
```json
{
"tenant_id": "default",
"agent_id": "telecom_contas",
"profile": "judge",
"operation": "judge.response_quality",
"messages": [
{"role": "system", "content": "Você é um avaliador."},
{"role": "user", "content": "Avalie a resposta."}
],
"metadata": {
"request_id": "req-001",
"trace_id": "trace-001"
}
}
- standardized RAG retrieval;
- cache key for LLM calls;
- hit/miss with telemetry;
- distributed cache through `create_cache(settings)`.
### 5. Unit tests
The following directory was created:
```text
tests/unit
```
### 11. LLMResponse
Initial coverage:
```json
{
"provider": "oci_openai",
"model": "openai.gpt-4.1",
"profile": "judge",
"content": "Resultado",
"usage": {
"input_tokens": 1200,
"output_tokens": 300,
"total_tokens": 1500
},
"latency_ms": 820
}
- cache;
- SSE;
- RAG;
- checkpoint saver;
- LangGraph telemetry;
- agent runtime;
- static workflow verification;
- main imports.
Local validation performed:
```text
12 passed
```
### 12. EvaluationRun
### How to test
```json
{
"run_id": "eval-001",
"agent_id": "telecom_contas",
"source": "langfuse",
"period_start": "2026-06-18T00:00:00Z",
"period_end": "2026-06-19T00:00:00Z",
"status": "running"
}
```bash
cd projeto_agent_framework_first_ready
pip install -r agent_template_backend/requirements.txt
pytest -q tests/unit
```
### 13. EventEnvelope
### Source files
```json
{
"event_type": "IC.AGENT_COMPLETED",
"timestamp": "2026-06-19T12:00:00Z",
"tenant_id": "default",
"agent_id": "telecom_contas",
"session_id": "session-001",
"trace_id": "trace-001",
"payload": {}
}
```
The files below were consolidated into this manual:
### 14. Regras de evolução
- `docs/RAG_PROVIDER_KBDB.md`
- `docs/README_rag_samples.md`
- `Documentacao/README_TEMPLATE_BUSINESS_CONTEXT_V2.md`
- `Documentacao/README_FIRST_MAX_OPERATIONAL_FIXES.md`
- campos novos devem ser opcionais;
- campos obrigatórios não podem ser removidos dentro da mesma major;
- mudança semântica exige nova versão;
- contratos são versionados independentemente.
### Maintenance rule
### 15. Erros comuns
| Erro | Impacto | Correção |
| --- | --- | --- |
| Payload bruto no Runtime | Acopla canais ao core. | Usar GatewayRequest. |
| Tool recebendo BusinessContext bruto sem mapping | Quebra contrato da tool. | Usar mcp_parameter_mapping.yaml. |
| LLM direto no agente | Quebra AI Gateway. | Usar LLMRequest/profile. |
| Campos sem versão | Dificulta migração. | Declarar contract_version. |
### 16. Critérios de aceite
- [ ] GatewayRequest documentado e versionado.
- [ ] ChannelResponse documentado e versionado.
- [ ] BusinessContext usado por canais e MCP.
- [ ] ToolInvocation e ToolResult padronizados.
- [ ] LLMRequest e LLMResponse padronizados.
- [ ] EvaluationRun e EvaluationResult padronizados.
- [ ] EventEnvelope usado para IC/NOC/GRL.
- [ ] Contratos possuem regras de evolução.
New fixes or evolutions for this subject should update this consolidated document. Release notes may continue to exist as history, but they should not be required to understand or implement the feature.

File diff suppressed because it is too large Load Diff

View File

@@ -1,506 +1,117 @@
### LLM Rich Response and reasoning_content
### How to use this manual
This is a **specialized reference manual**. It does not replace the main tutorial.
- To build an agent end to end, use [`README_en.md`](../../../README_en.md).
- Use this document when implementing, deep-diving or troubleshooting **`ainvoke_response()`, inference metadata and optional `reasoning_content`**.
- Historical examples consolidated here must be interpreted against the current framework API.
- If documentation differs, the current code and root README take precedence.
- To create an agent from start to finish, use [`README_en.md`](../../../README_en.md).
- Use this document when you need to implement, deepen, or diagnose **`ainvoke_response()`, inference metadata, and optional `reasoning_content`**.
- Historical examples consolidated here should be read in light of the framework's current API.
- In case of divergence, the code for the version and the current `README_en.md` take precedence.
### Relationship with the main tutorial
`README_en.md` introduces this capability as part of the normal development flow. This manual consolidates details previously spread across `docs/`, `Documentacao/`, release notes, validation records and specialized guides.
The `README_en.md` presents this capability in the normal development flow. This manual brings together details that were distributed across `docs/`, `Documentacao/`, release notes, validations, and specialized guides.
Its purpose is to answer **“how does this feature work in depth and how do I troubleshoot it?”** without becoming a second copy of the main tutorial.
The goal here is to answer **“how does this feature work in depth and how do I solve problems with it?”**, without turning this file into a second copy of the main tutorial.
### Scope
`ainvoke_response()`, inference metadata and optional `reasoning_content`.
`ainvoke_response()`, inference metadata, and optional `reasoning_content`.
### Consolidated technical content
### LLM Rich Response and reasoning_content
The LLM abstraction keeps the legacy string-returning API and adds an opt-in structured response for consumers that need inference metadata.
Guide for using the opt-in structured LLM response API without breaking the legacy `ainvoke()` contract, including `reasoning_content`, usage, model, provider, fallback, and tests.
### Legacy API
### How to use this document
`ainvoke()` continues to return `str`. Existing agents do not need to change and callers that do not need metadata should keep using it.
This is the consolidated development document for this subject. It brings together architecture, configuration, examples, runtime behavior, compatibility, tests, and troubleshooting that were previously distributed across several files. Source sections were preserved when they provided distinct technical details; release notes were incorporated as current behavior or correction history.
### Rich API
### Rich LLM response API
`ainvoke_response()` returns a structured object containing the final content and, when available, `reasoning_content`, usage, model and provider metadata.
> Content consolidated from `docs/LLM_RICH_RESPONSE.md`.
`reasoning_content` is optional. The framework never fabricates it. If a provider/model does not expose this field, the value is `None`. The reasoning field remains separate from final user-visible content.
### Goal
### Backoffice use
The framework keeps `ainvoke()` as the backward-compatible API, returning only `str`, and adds `ainvoke_response()` for consumers that need additional inference metadata, including `reasoning_content` when the model/provider/API makes it available.
A Backoffice consumer that needs model-decision metadata may opt into `ainvoke_response()` while agent runtime paths that only need final content keep using `ainvoke()`.
### APIs
### Provider compatibility
### Legacy API — unchanged
Custom providers that only implement the legacy method continue to work through fallback behavior: the framework wraps the returned text as rich content and leaves reasoning metadata unset. Provider implementations that support richer metadata can override/implement the rich path directly.
```python
answer = await llm.ainvoke(messages)
assert isinstance(answer, str)
```
### Testing
No existing agent needs to be changed.
Cover legacy return type, provider with reasoning, provider without reasoning, fallback custom provider, usage/model/provider metadata and failure behavior.
### New rich API — opt-in
### Source material consolidated
```python
response = await llm.ainvoke_response(messages)
answer = response.content
reasoning = response.reasoning_content
usage = response.usage
model = response.model
provider = response.provider
```
`reasoning_content` is `str | None`. `None` is the expected behavior when the model, provider, or API does not expose textual reasoning.
### Backoffice
A consumer that previously did:
```python
answer = await llm.ainvoke(messages)
template = extract_response(answer)
```
can instead do:
```python
response = await llm.ainvoke_response(messages)
template = extract_response(response.content)
reasoning_content = response.reasoning_content
```
Logic that expects text continues to receive `response.content`; reasoning remains separate and does not contaminate response, cache, memory, judges, or guardrails.
### Custom-provider compatibility
`LLMProvider.ainvoke_response()` has a fallback. An external provider that implements only `ainvoke()` continues to work and automatically receives `LLMResponse(content=<texto>)`, with `reasoning_content=None`.
Native providers (`mock`, OpenAI-compatible/OCI OpenAI, and OCI SDK) implement the rich response and attempt to preserve reasoning when present.
### Compatibility guarantees
- `ainvoke()` continues to return `str`.
- No existing router, judge, RAG, memory, cache, or runtime has been migrated to the new API.
- `reasoning_content` is never fabricated by the framework.
- Missing reasoning does not generate an error.
- Existing telemetry output continues to be the final content, without automatically appending reasoning.
### Tests
Specific tests are in `tests/unit/test_llm_rich_response.py` and verify:
1. a legacy provider that implements only `ainvoke()`;
2. preservation of the `str` return from `ainvoke()`;
3. `LLMResponse` return from `ainvoke_response()`;
4. reasoning through a direct attribute;
5. reasoning through `model_extra`;
6. missing reasoning and extraction in OCI SDK format.
### Source files
The files below were consolidated into this manual:
- `docs/LLM_RICH_RESPONSE.md`
### Detailed normative and implementation reference
### Maintenance rule
The sections below preserve the detailed English project specifications and implementation guides relevant to this capability. They are included here so a developer does not need to reconstruct the behavior from separate documents.
### LLM runtime contract context
> Consolidated from `specs/SPEC-002-Agent-Runtime.md`.
### Escopo
O Agent Runtime executa o ciclo de vida conversacional do agente. A execução inclui normalização de contexto, estado LangGraph, memória, checkpoint, roteamento, supervisor, guardrails, MCP, RAG, LLM, judges, persistência e resposta final.
### Componentes
| Componente | Responsabilidade |
|---|---|
| Workflow Builder | Compila o grafo LangGraph. |
| State Manager | Mantém o estado de execução. |
| Session Manager | Resolve sessão e conversation_key. |
| Memory Manager | Carrega e persiste histórico. |
| Checkpoint Manager | Persiste estado LangGraph. |
| Input Guardrail Node | Executa guardrails de entrada. |
| Router Node | Decide rota/intent. |
| Supervisor Node | Decide handoff ou próximo agente quando habilitado. |
| Agent Node | Executa agente de domínio. |
| MCP Client/Router | Executa tools por contrato. |
| RAG Service | Recupera contexto documental. |
| Output Supervisor | Revisa resposta antes de saída. |
| Output Guardrail Node | Executa guardrails de saída. |
| Judge Node | Avalia resposta. |
| Persistence Node | Persiste mensagens, memória e checkpoint. |
### State Model
```python
class AgentState(TypedDict, total=False):
user_text: str
sanitized_input: str
response_text: str
tenant_id: str
agent_id: str
channel: str
session_id: str
conversation_key: str
message_id: str
route: str
intent: str
context: dict
business_context: dict
tool_arguments: dict
mcp_tools: list[str]
mcp_results: list[dict]
rag_context: str
rag_metadata: dict
guardrails: list[dict]
judges: list[dict]
metadata: dict
errors: list[dict]
```
### Workflow
```mermaid
flowchart TD
A[start] --> B[input_guardrails]
B --> C[routing_decision]
C --> D[agent_execution]
D --> E[output_supervisor]
E --> F[output_guardrails]
F --> G[judge]
G --> H[persist]
H --> I[end]
C --> J[handoff]
J --> C
```
### Nós
| Nó | Entrada | Saída |
|---|---|---|
| `input_guardrails` | `user_text`, `context` | `sanitized_input`, `guardrails` |
| `routing_decision` | `sanitized_input`, `business_context` | `route`, `intent`, `mcp_tools` |
| `agent_execution` | `state` completo | `response_text`, `mcp_results`, `rag_metadata` |
| `output_supervisor` | `response_text` | `response_text` revisado |
| `output_guardrails` | `response_text` | `response_text`, `guardrails` |
| `judge` | `response_text`, evidências | `judges` |
| `persist` | `state` completo | checkpoint, memória, mensagens |
### Router
```yaml
routing:
mode: router
fallback_agent: billing_agent
enable_llm_router: false
intents:
billing_invoice_explanation:
route: billing_agent
keywords:
- fatura
- cobrança
- boleto
mcp_tools:
- consultar_fatura
- consultar_pagamentos
```
### Supervisor
```yaml
supervisor:
enabled: true
profile: supervisor
max_turns: 5
handoff_enabled: true
fallback_route: support_agent
```
### Memory
| Provider | Uso |
|---|---|
| `memory` | Execução local e testes. |
| `sqlite` | Desenvolvimento local persistente. |
| `mongodb` | Checkpoint e histórico em ambiente distribuído. |
| `autonomous` | Produção com Oracle Autonomous Database. |
### Checkpoints
Checkpoint contém:
```json
{
"conversation_key": "default:telecom_contas:session-001",
"checkpoint_id": "ckpt-001",
"state": {},
"pending_writes": [],
"created_at": "2026-06-19T12:00:00Z"
}
```
Formato entregue ao LangGraph:
```python
pending_writes: list[tuple[str, str, object]]
```
### Business Context
```yaml
business_context:
customer_key: "11999999999"
contract_key: "3000131180"
interaction_key: "301953872"
account_key: null
resource_key: null
session_key: "session-001"
metadata:
source_channel: web
```
### Ordem de Prioridade dos Dados
1. `tool_arguments`
2. `business_context`
3. `context`
4. `session.metadata`
5. `state`
6. extração complementar do texto
### MCP Integration
```mermaid
flowchart LR
AgentNode --> ToolList[mcp_tools]
ToolList --> Mapping[mcp_parameter_mapping.yaml]
Mapping --> MCP[MCP Gateway/Router]
MCP --> Result[mcp_results]
```
### RAG Integration
```yaml
rag:
enabled: true
namespace_strategy: agent_id
top_k: 5
profile_generation: rag_generation
```
### Eventos
| Evento | Descrição |
|---|---|
| `runtime.started` | Execução iniciada. |
| `runtime.session.loaded` | Sessão carregada. |
| `runtime.memory.loaded` | Memória carregada. |
| `runtime.checkpoint.loaded` | Checkpoint carregado. |
| `runtime.route.selected` | Rota selecionada. |
| `runtime.agent.started` | Agente iniciado. |
| `runtime.agent.completed` | Agente concluído. |
| `runtime.persist.completed` | Persistência concluída. |
| `runtime.failed` | Falha controlada. |
### Erros
| Código | Condição | Tratamento |
|---|---|---|
| `RUNTIME_INVALID_REQUEST` | GatewayRequest inválido | 422 |
| `RUNTIME_ROUTE_NOT_FOUND` | Nenhuma rota elegível | fallback ou resposta controlada |
| `RUNTIME_CHECKPOINT_ERROR` | Falha em checkpoint | retry ou stateless conforme config |
| `RUNTIME_MEMORY_ERROR` | Falha em memória | retry ou resposta controlada |
| `RUNTIME_AGENT_ERROR` | Falha no agente | NOC + fallback |
| `RUNTIME_TIMEOUT` | Timeout geral | resposta controlada |
### Contrato Durável de Estado Transacional
Hosts que utilizam `AgentRuntime` com transações multi-turno DEVEM declarar no `AgentState` os campos `active_transaction` e `last_transaction`. O primeiro é a fonte canônica da transação em andamento e deve sobreviver a checkpoint/resume; o segundo mantém o snapshot da última transação terminal.
```python
active_transaction: dict[str, Any]
last_transaction: dict[str, Any]
```
`selected_tool_call` e `pending_tool_call` são campos auxiliares/compatibilidade e não substituem o latch canônico. Durante `COLLECTING_PARAMETERS`, a retomada da transação e o consumo de parâmetros pendentes têm precedência sobre keyword routing genérico. Uma mudança de intenção só deve interromper a transação quando for inequívoca ou explicitamente solicitada pelo usuário.
O contrato completo, ciclo de vida, precedência de roteamento, checklist e testes regressivos estão em [`docs/TRANSACTION_STATE_DEVELOPER_GUIDE.md`](../docs/TRANSACTION_STATE_DEVELOPER_GUIDE.md).
### Requisitos Não Funcionais
| Categoria | Requisito |
|---|---|
| Disponibilidade | Componentes deployáveis expõem `/health` e `/ready`. |
| Escalabilidade | Apps stateless escalam horizontalmente. Estado conversacional fica em repositórios externos. |
| Segurança | Segredos são fornecidos por secret store ou Kubernetes Secrets. |
| Observabilidade | Logs, métricas e traces usam correlação por request_id, trace_id, session_id, tenant_id e agent_id. |
| Auditabilidade | Decisões de rota, guardrail, judge, MCP e LLM são rastreáveis. |
| Portabilidade | Execução suportada em local, Docker Compose e Kubernetes/OKE. |
| Configuração | Comportamento variável é controlado por `.env` e YAML versionado. |
### Critérios de Aceite
- [ ] Runtime recebe GatewayRequest validado.
- [ ] State contém tenant_id, agent_id, session_id, conversation_key, route e intent.
- [ ] Input guardrails executam antes do roteamento.
- [ ] Router ou Supervisor seleciona rota.
- [ ] Agent Node executa sem acessar payload bruto de canal.
- [ ] MCP é acessado por contrato.
- [ ] RAG é acessado por serviço reutilizável.
- [ ] Output guardrails executam antes da resposta final.
- [ ] Judges geram JudgeResult.
- [ ] Memória e checkpoint são persistidos conforme provider.
- [ ] Hosts transacionais declaram `active_transaction` e `last_transaction` no `AgentState`.
- [ ] Durante `COLLECTING_PARAMETERS`, respostas a parâmetros pendentes têm precedência sobre keyword routing genérico.
- [ ] Erros geram NOC e resposta controlada.
### Glossário
| Termo | Definição |
|---|---|
| Agent Platform | Plataforma composta por runtime, gateways, evaluator, templates, contratos e componentes operacionais. |
| Agent Framework | Biblioteca/core reutilizável com contratos, guardrails, judges, memória, telemetria, providers e utilitários. |
| Agent Runtime | Motor de execução de agentes baseado em LangGraph, estado, sessão, memória, checkpoints, roteamento e ciclo de vida. |
| Agent Gateway | Aplicação deployável de entrada, roteamento e orquestração entre backends/agentes. |
| Channel Gateway | Aplicação ou módulo de normalização de payloads de canais para GatewayRequest. |
| AI Gateway | Aplicação de governança, roteamento e abstração de chamadas LLM/embedding. |
| MCP Gateway | Aplicação de governança e roteamento de tools MCP. |
| Evaluator | Camada de avaliação online/offline, regressão e certificação. |
| Business Context | Conjunto de chaves canônicas de negócio: customer_key, contract_key, interaction_key, account_key, resource_key e session_key. |
### Compatibility rules
> Consolidated from `specs/SPEC-013-Versioning-and-Compatibility-Model.md`.
### Agent Platform OCI
Version: 1.0.0
---
### Padrão de leitura
Cada SPEC está organizada para servir tanto como contrato arquitetural quanto como guia prático de adoção.
A estrutura usada é:
1. Conceito.
2. Problema que resolve.
3. Quando usar.
4. Quando não usar.
5. Arquitetura.
6. Implementação.
7. Exemplos.
8. Erros comuns.
9. Critérios de aceite.
---
### 1. Conceito
Versionamento define como a plataforma evolui sem quebrar projetos existentes. Compatibilidade define quais versões de framework, runtime, gateways, contracts, templates, prompts, tools e evaluator podem operar juntas.
### 2. Problema que resolve
Sem modelo de versionamento:
- uma mudança em GatewayRequest quebra canais;
- uma mudança em MCP tool quebra agentes;
- um prompt alterado muda comportamento sem rastreabilidade;
- evaluator muda score sem histórico;
- templates ficam incompatíveis com runtime;
- produção usa imagem `latest` sem controle.
### 3. Semantic Versioning
Formato:
```text
MAJOR.MINOR.PATCH
```
Regras:
| Parte | Significado |
| --- | --- |
| MAJOR | Mudança incompatível. |
| MINOR | Nova capacidade compatível. |
| PATCH | Correção sem mudança de contrato. |
### 4. Artefatos versionados
| Artefato | Modelo |
| --- | --- |
| agent_framework | SemVer |
| agent_runtime | SemVer alinhado ao framework |
| agent_gateway | SemVer + Docker tag |
| channel_gateway | SemVer + Docker tag |
| ai_gateway | SemVer + Docker tag |
| mcp_gateway | SemVer + Docker tag |
| templates | versão da plataforma |
| contracts | contract-name-vN |
| prompts | SemVer |
| datasets | SemVer |
| guardrails | SemVer por código |
| judges | SemVer por judge |
| mcp_tools | SemVer por tool |
| evaluator | SemVer |
| certification_suite | SemVer + ruleset version |
### 5. Contract versioning
Exemplos:
```text
gateway-request-v1
business-context-v1
tool-invocation-v1
llm-request-v1
```
Permitido na mesma versão major:
- adicionar campos opcionais;
- adicionar metadata;
- adicionar enum documentado.
Não permitido:
- remover campo obrigatório;
- mudar tipo;
- mudar significado;
- alterar regra obrigatória.
### 6. Compatibility Matrix
```yaml
compatibility:
- framework: "1.4.x"
runtime: "1.4.x"
agent_gateway: "1.4.x"
supported: true
- framework: "1.4.x"
runtime: "2.0.x"
supported: false
```
### 7. Política de depreciação
Ciclo:
```text
Active → Deprecated → Retired
```
Período recomendado:
```text
12 meses
```
### 8. Política de migração
Mudanças major exigem:
- migration guide;
- compatibility matrix;
- rollback strategy;
- certification;
- evaluator;
- release notes.
### 9. Estratégia de rollback
Rollback deve considerar:
- imagem Docker;
- versão do pacote;
- versão dos YAMLs;
- versão do contrato;
- migration de banco;
- dataset;
- prompts.
### 10. Erros comuns
| Erro | Impacto | Correção |
| --- | --- | --- |
| Usar latest em produção | Deploy não reprodutível. | Usar tag explícita. |
| Mudar prompt sem versão | Sem rastreabilidade. | Versionar prompt. |
| Adicionar campo obrigatório em contrato v1 | Quebra clientes. | Criar v2. |
| Atualizar evaluator sem baseline | Scores não comparáveis. | Registrar versão e metodologia. |
### 11. Critérios de aceite
- [ ] Todos os componentes têm versão.
- [ ] Contratos têm versão independente.
- [ ] Matriz de compatibilidade publicada.
- [ ] Release notes publicadas.
- [ ] Migrações major possuem guide.
- [ ] Rollback definido.
- [ ] Prompts e datasets versionados.
- [ ] Evaluator e certification registram versão.
New fixes or evolutions for this subject should update this consolidated document. Release notes may continue to exist as history, but they should not be required to understand or implement the feature.

View File

@@ -1,499 +1,360 @@
### Performance, Cache and Async Runtime
### Performance, Cache, and Async Runtime
### How to use this manual
This is a **specialized reference manual**. It does not replace the main tutorial.
- To build an agent end to end, use [`README_en.md`](../../../README_en.md).
- Use this document when implementing, deep-diving or troubleshooting **concurrency, caching, reduction of LLM calls and cross-loop fixes**.
- Historical examples consolidated here must be interpreted against the current framework API.
- If documentation differs, the current code and root README take precedence.
- To create an agent from start to finish, use [`README_en.md`](../../../README_en.md).
- Use this document when you need to implement, deepen, or diagnose **concurrency, cache, reduction of LLM calls, and cross-loop fixes**.
- Historical examples consolidated here should be read in light of the framework's current API.
- In case of divergence, the code for the version and the current `README_en.md` take precedence.
### Relationship with the main tutorial
`README_en.md` introduces this capability as part of the normal development flow. This manual consolidates details previously spread across `docs/`, `Documentacao/`, release notes, validation records and specialized guides.
The `README_en.md` presents this capability in the normal development flow. This manual brings together details that were distributed across `docs/`, `Documentacao/`, release notes, validations, and specialized guides.
Its purpose is to answer **“how does this feature work in depth and how do I troubleshoot it?”** without becoming a second copy of the main tutorial.
The goal here is to answer **“how does this feature work in depth and how do I solve problems with it?”**, without turning this file into a second copy of the main tutorial.
### Scope
Concurrency, caching, reduction of llm calls and cross-loop fixes.
Concurrency, cache, reduction of LLM calls, and cross-loop fixes.
### Consolidated technical content
### Performance, Cache, Concurrency and Asynchronous Runtime
### Performance, Cache, Concurrency, and Async Runtime
This guide collects optimizations that reduce latency without changing functional semantics.
Manual for optimizations on the critical MCP, RAG, and Judge path, reduction of LLM calls, deterministic preemption, and cross-loop deadlock correction in sequencing.
### Optimization principles
### How to use this document
Use deterministic signals before expensive semantic calls when they are reliable; execute independent work concurrently; avoid recomputing retrieval/tool metadata; cache only when correctness allows it; and keep I/O asynchronous without sharing loop-bound primitives incorrectly.
This is the consolidated development document for this subject. It brings together architecture, configuration, examples, runtime behavior, compatibility, tests, and troubleshooting that were previously distributed across several files. Source sections were preserved when they provided distinct technical details; release notes were incorporated as current behavior or correction history.
### MCP/RAG/Judges
### MCP, RAG, and Judge optimizations
MCP preparation and repeated metadata operations can be reused where safe. RAG should avoid repeated retrieval/embedding work through configured cache layers. Independent judges can execute concurrently instead of serially.
> Content consolidated from `docs/PERFORMANCE_OPTIMIZATIONS_MCP_JUDGES_RAG.md`.
Transactional judge rules still override normal sampling optimization: performance must not skip critical evaluation.
- `mcp_tools` remains an allowlist; only the query selected through `selection_keywords` is executed.
- `strategy: hybrid` extraction tries a regex `pattern` before the LLM profile.
- RAG is skipped when successful MCP evidence is sufficient, except for policy/rule questions.
- `mcp_results` is provided as evidence to the groundedness judge.
- `judges.yaml` accepts `sample_rate` and `always_run_for_transactional`.
- Simple structured queries can return a deterministic response without invoking the agent LLM.
### Routing optimization
### Shift from query to transactional action
Explicit intent-shift signals can preempt the route-continuity LLM. This reduces token consumption and latency while preserving semantic fallback for ambiguous cases.
Route stickiness is preempted when an explicit keyword configured in `routing.yaml` identifies another intent/agent. Thus, a session in `retail_order_tracking` moves to `retail_support_exchange_return` when it receives requests such as “return order”. In addition, direct responses from read-only tools are blocked when the message contains `selection_keywords` from any registered transactional tool.
Action words remain in `config/tools.yaml`; the runtime does not maintain hardcoded domain aliases.
### Deterministic preemption for an explicit intent change
Stickiness does not call a second LLM when the message contains an explicit change that can be recognized deterministically. Multi-token keywords configured in `routing.yaml` accept up to three intermediate tokens while preserving order. Therefore, `cancelar pedido` recognizes `quero cancelar meu pedido`, `cancelar o meu pedido`, and `pode cancelar esse pedido`. In this case the new intent preempts stickiness and the `keyword_match_strategy=ordered_tokens` metadata makes the decision auditable. Messages with no explicit signal continue using route stickiness normally.
### Cross-loop deadlock fix
Sequence generation/observability previously could wait on synchronization primitives associated with another event loop. The fix removes cross-loop waiting and keeps sequencing safe for asynchronous runtime and tests that create multiple loops.
> Content consolidated from `Documentacao/FIX_DEADLOCK_SEQUENCE_CROSS_LOOP.md`.
### Validation
### Problem
Performance tests should measure latency and call counts, not only functional output. Regression coverage should include concurrent judges, cached/uncached RAG behavior, MCP reuse paths, deterministic routing preemption and observer/sequence calls across separate event loops.
The synchronous `agent_framework.observer.event()` API could be called from a worker thread with no active event loop. In that case, the previous implementation ran `asyncio.run(aevent(...))`, creating a temporary new event loop. At the same time, `analytics/tim_sequence.py` shared global `asyncio.Lock` instances (`_mongo_index_lock` and `_memory_lock`) across calls that could come from different event loops.
### Source material consolidated
On the first Mongo operation, `_ensure_mongo_ttl_index_once()` held `_mongo_index_lock` while creating the TTL index. Contention from another loop could leave the second call waiting indefinitely.
### Applied changes
1. `observer.py`
- removed `asyncio.run()` from the synchronous `event()` path;
- added a dedicated reusable event loop for synchronous calls;
- cross-thread submission uses `asyncio.run_coroutine_threadsafe()`;
- best-effort loop shutdown when the process terminates.
2. `analytics/tim_sequence.py`
- `_mongo_index_lock`: `asyncio.Lock` -> `threading.Lock`;
- `_memory_lock`: `asyncio.Lock` -> `threading.Lock`;
- TTL-index initialization moved to a synchronous function protected by a thread lock and called through `asyncio.to_thread()`;
- the in-memory fallback counter uses a short thread-safe critical section.
3. Tests
- `tests/test_observer_cross_loop_deadlock_fix.py` validates:
- multiple worker threads using `event()` share the same synchronous observer loop;
- in-memory sequence remains monotonic across independent event loops;
- TTL-index creation happens only once under cross-loop contention.
### Validation performed
```bash
PYTHONPATH=libs/agent_framework/src pytest -q tests/test_observer_cross_loop_deadlock_fix.py
```
Result: `3 passed`.
The full repository suite has pre-existing/independent failures unrelated to this change, including collection conflicts for `test_long_term_memory.py`, static template paths, and checkpoint/workflow tests. Those items were not changed by this fix.
### Operational performance features
> Content consolidated from `Documentacao/README_MAX_OPERACIONAL.md`.
This version adds the operational adjustments that were missing to bring the framework closer to the FIRST production standard.
### Adjustments included in this version
### 1. Langfuse Enterprise Adapter
New module:
```text
agent_framework/observability/langfuse_enterprise.py
```
Includes an adapter compatible with Langfuse SDKs v2/v3 for:
- trace updates;
- trace scoring/evaluation;
- prompt registry when supported by the SDK;
- isolation of Langfuse API differences.
### 2. Persistent Token and Cost Accounting
New package:
```text
agent_framework/billing/
```
Includes:
- `UsageRecord`
- `SQLiteUsageRepository`
- `OracleUsageRepository`
- `create_usage_repository(settings)`
The LLM provider now records automatically:
- `prompt_tokens`
- `completion_tokens`
- `cached_tokens`
- `total_tokens`
- `cost_usd`
- `cost_brl`
- `tenant_id`
- `agent_id`
- `session_id`
- `message_id`
New endpoint:
```http
GET /debug/usage
GET /debug/usage?tenant_id=default
GET /debug/usage?session_id=<id>
```
### 3. Operational RAG Service
New module:
```text
agent_framework/rag/rag_service.py
```
Includes:
- `RagService.add_documents()`
- `RagService.retrieve()`
- `RagResult.as_prompt_context()`
- telemetry for latency, document count, top scores, and graph.
### 4. New configuration
Variable added:
```env
USAGE_REPOSITORY_PROVIDER=sqlite
```
Values:
```text
sqlite
oracle
autonomous
```
### 5. Local operational compatibility
By default, usage accounting uses SQLite even when everything else is in memory. This makes local testing possible without Oracle.
### Quick test
```bash
cd agent_template_backend
uvicorn app.main:app --host 0.0.0.0 --port 8000
```
Test a message:
```bash
curl -X POST http://localhost:8000/gateway/message \
-H 'Content-Type: application/json' \
-d '{"channel":"web","payload":{"text":"teste","user_id":"u1","session_id":"s1"}}'
```
Check usage/cost:
```bash
curl http://localhost:8000/debug/usage
```
### To run closer to a production pattern
```env
SESSION_REPOSITORY_PROVIDER=sqlite
MEMORY_REPOSITORY_PROVIDER=sqlite
CHECKPOINT_REPOSITORY_PROVIDER=sqlite
USAGE_REPOSITORY_PROVIDER=sqlite
CACHE_BACKEND_PROVIDER=sqlite
VECTOR_STORE_PROVIDER=sqlite
ENABLE_LANGFUSE=true
LANGFUSE_HOST=http://localhost:3000
LANGFUSE_PUBLIC_KEY=...
LANGFUSE_SECRET_KEY=...
```
For Autonomous Database:
```env
SESSION_REPOSITORY_PROVIDER=oracle
MEMORY_REPOSITORY_PROVIDER=oracle
CHECKPOINT_REPOSITORY_PROVIDER=oracle
USAGE_REPOSITORY_PROVIDER=oracle
CACHE_BACKEND_PROVIDER=oracle
VECTOR_STORE_PROVIDER=oracle
GRAPH_STORE_PROVIDER=oracle
ADB_USER=...
ADB_PASSWORD=...
ADB_DSN=...
ADB_WALLET_LOCATION=...
ADB_TABLE_PREFIX=AGENTFW
```
### Final cache, RAG, and telemetry adjustments
> Content consolidated from `Documentacao/README_FIRST_MAX_OPERATIONAL_FIXES.md`.
This version fixes the gaps identified in the comparison against FIRST.
### Applied fixes
### 1. Operational LangGraph checkpoint
The workflow no longer compiles directly with `MemorySaver()`. The following adapter was created:
```text
agent_framework/checkpoints/langgraph_saver.py
```
It connects LangGraph to the framework's configured repository:
- `memory`
- `sqlite`
- `oracle` / `autonomous`
In the workflow:
```python
builder.compile(checkpointer=create_langgraph_checkpointer(self.settings))
```
### 2. LangGraph telemetry wrapping actual execution
A node wrapper was added to the workflow:
```python
self._node("billing_agent", self.billing_agent)
```
This way the `langgraph.node.*` span/event wraps actual node execution, not just an empty block.
Events emitted:
- `langgraph.node.started`
- `langgraph.node.completed`
- `langgraph.node.failed`
- `langgraph.edge.selected`
### 3. RAG integrated into agents
Agents now receive `RagService` and use retrieved context in the prompt:
- BillingAgent
- ProductAgent
- OrdersAgent
- SupportAgent
RAG uses:
- `VECTOR_STORE_PROVIDER=memory|sqlite|oracle|autonomous`
- `GRAPH_STORE_PROVIDER=memory|oracle|autonomous`
- `RAG_TOP_K`
### 4. Cache integrated into agent runtime
The following mixin was created:
```text
agent_template_backend/app/agents/runtime.py
```
It adds:
- standardized RAG retrieval;
- cache key for LLM calls;
- hit/miss with telemetry;
- distributed cache through `create_cache(settings)`.
### 5. Unit tests
The following directory was created:
```text
tests/unit
```
Initial coverage:
- cache;
- SSE;
- RAG;
- checkpoint saver;
- LangGraph telemetry;
- agent runtime;
- static workflow verification;
- main imports.
Local validation performed:
```text
12 passed
```
### How to test
```bash
cd projeto_agent_framework_first_ready
pip install -r agent_template_backend/requirements.txt
pytest -q tests/unit
```
### Source files
The files below were consolidated into this manual:
- `docs/PERFORMANCE_OPTIMIZATIONS_MCP_JUDGES_RAG.md`
- `Documentacao/FIX_DEADLOCK_SEQUENCE_CROSS_LOOP.md`
- operational notes in `Documentacao/README_MAX_OPERACIONAL.md` and `README_FIRST_MAX_OPERATIONAL_FIXES.md`
- `Documentacao/README_MAX_OPERACIONAL.md`
- `Documentacao/README_FIRST_MAX_OPERATIONAL_FIXES.md`
### Detailed normative and implementation reference
### Maintenance rule
The sections below preserve the detailed English project specifications and implementation guides relevant to this capability. They are included here so a developer does not need to reconstruct the behavior from separate documents.
### Runtime execution requirements
> Consolidated from `specs/SPEC-002-Agent-Runtime.md`.
### Escopo
O Agent Runtime executa o ciclo de vida conversacional do agente. A execução inclui normalização de contexto, estado LangGraph, memória, checkpoint, roteamento, supervisor, guardrails, MCP, RAG, LLM, judges, persistência e resposta final.
### Componentes
| Componente | Responsabilidade |
|---|---|
| Workflow Builder | Compila o grafo LangGraph. |
| State Manager | Mantém o estado de execução. |
| Session Manager | Resolve sessão e conversation_key. |
| Memory Manager | Carrega e persiste histórico. |
| Checkpoint Manager | Persiste estado LangGraph. |
| Input Guardrail Node | Executa guardrails de entrada. |
| Router Node | Decide rota/intent. |
| Supervisor Node | Decide handoff ou próximo agente quando habilitado. |
| Agent Node | Executa agente de domínio. |
| MCP Client/Router | Executa tools por contrato. |
| RAG Service | Recupera contexto documental. |
| Output Supervisor | Revisa resposta antes de saída. |
| Output Guardrail Node | Executa guardrails de saída. |
| Judge Node | Avalia resposta. |
| Persistence Node | Persiste mensagens, memória e checkpoint. |
### State Model
```python
class AgentState(TypedDict, total=False):
user_text: str
sanitized_input: str
response_text: str
tenant_id: str
agent_id: str
channel: str
session_id: str
conversation_key: str
message_id: str
route: str
intent: str
context: dict
business_context: dict
tool_arguments: dict
mcp_tools: list[str]
mcp_results: list[dict]
rag_context: str
rag_metadata: dict
guardrails: list[dict]
judges: list[dict]
metadata: dict
errors: list[dict]
```
### Workflow
```mermaid
flowchart TD
A[start] --> B[input_guardrails]
B --> C[routing_decision]
C --> D[agent_execution]
D --> E[output_supervisor]
E --> F[output_guardrails]
F --> G[judge]
G --> H[persist]
H --> I[end]
C --> J[handoff]
J --> C
```
### Nós
| Nó | Entrada | Saída |
|---|---|---|
| `input_guardrails` | `user_text`, `context` | `sanitized_input`, `guardrails` |
| `routing_decision` | `sanitized_input`, `business_context` | `route`, `intent`, `mcp_tools` |
| `agent_execution` | `state` completo | `response_text`, `mcp_results`, `rag_metadata` |
| `output_supervisor` | `response_text` | `response_text` revisado |
| `output_guardrails` | `response_text` | `response_text`, `guardrails` |
| `judge` | `response_text`, evidências | `judges` |
| `persist` | `state` completo | checkpoint, memória, mensagens |
### Router
```yaml
routing:
mode: router
fallback_agent: billing_agent
enable_llm_router: false
intents:
billing_invoice_explanation:
route: billing_agent
keywords:
- fatura
- cobrança
- boleto
mcp_tools:
- consultar_fatura
- consultar_pagamentos
```
### Supervisor
```yaml
supervisor:
enabled: true
profile: supervisor
max_turns: 5
handoff_enabled: true
fallback_route: support_agent
```
### Memory
| Provider | Uso |
|---|---|
| `memory` | Execução local e testes. |
| `sqlite` | Desenvolvimento local persistente. |
| `mongodb` | Checkpoint e histórico em ambiente distribuído. |
| `autonomous` | Produção com Oracle Autonomous Database. |
### Checkpoints
Checkpoint contém:
```json
{
"conversation_key": "default:telecom_contas:session-001",
"checkpoint_id": "ckpt-001",
"state": {},
"pending_writes": [],
"created_at": "2026-06-19T12:00:00Z"
}
```
Formato entregue ao LangGraph:
```python
pending_writes: list[tuple[str, str, object]]
```
### Business Context
```yaml
business_context:
customer_key: "11999999999"
contract_key: "3000131180"
interaction_key: "301953872"
account_key: null
resource_key: null
session_key: "session-001"
metadata:
source_channel: web
```
### Ordem de Prioridade dos Dados
1. `tool_arguments`
2. `business_context`
3. `context`
4. `session.metadata`
5. `state`
6. extração complementar do texto
### MCP Integration
```mermaid
flowchart LR
AgentNode --> ToolList[mcp_tools]
ToolList --> Mapping[mcp_parameter_mapping.yaml]
Mapping --> MCP[MCP Gateway/Router]
MCP --> Result[mcp_results]
```
### RAG Integration
```yaml
rag:
enabled: true
namespace_strategy: agent_id
top_k: 5
profile_generation: rag_generation
```
### Eventos
| Evento | Descrição |
|---|---|
| `runtime.started` | Execução iniciada. |
| `runtime.session.loaded` | Sessão carregada. |
| `runtime.memory.loaded` | Memória carregada. |
| `runtime.checkpoint.loaded` | Checkpoint carregado. |
| `runtime.route.selected` | Rota selecionada. |
| `runtime.agent.started` | Agente iniciado. |
| `runtime.agent.completed` | Agente concluído. |
| `runtime.persist.completed` | Persistência concluída. |
| `runtime.failed` | Falha controlada. |
### Erros
| Código | Condição | Tratamento |
|---|---|---|
| `RUNTIME_INVALID_REQUEST` | GatewayRequest inválido | 422 |
| `RUNTIME_ROUTE_NOT_FOUND` | Nenhuma rota elegível | fallback ou resposta controlada |
| `RUNTIME_CHECKPOINT_ERROR` | Falha em checkpoint | retry ou stateless conforme config |
| `RUNTIME_MEMORY_ERROR` | Falha em memória | retry ou resposta controlada |
| `RUNTIME_AGENT_ERROR` | Falha no agente | NOC + fallback |
| `RUNTIME_TIMEOUT` | Timeout geral | resposta controlada |
### Contrato Durável de Estado Transacional
Hosts que utilizam `AgentRuntime` com transações multi-turno DEVEM declarar no `AgentState` os campos `active_transaction` e `last_transaction`. O primeiro é a fonte canônica da transação em andamento e deve sobreviver a checkpoint/resume; o segundo mantém o snapshot da última transação terminal.
```python
active_transaction: dict[str, Any]
last_transaction: dict[str, Any]
```
`selected_tool_call` e `pending_tool_call` são campos auxiliares/compatibilidade e não substituem o latch canônico. Durante `COLLECTING_PARAMETERS`, a retomada da transação e o consumo de parâmetros pendentes têm precedência sobre keyword routing genérico. Uma mudança de intenção só deve interromper a transação quando for inequívoca ou explicitamente solicitada pelo usuário.
O contrato completo, ciclo de vida, precedência de roteamento, checklist e testes regressivos estão em [`docs/TRANSACTION_STATE_DEVELOPER_GUIDE.md`](../docs/TRANSACTION_STATE_DEVELOPER_GUIDE.md).
### Requisitos Não Funcionais
| Categoria | Requisito |
|---|---|
| Disponibilidade | Componentes deployáveis expõem `/health` e `/ready`. |
| Escalabilidade | Apps stateless escalam horizontalmente. Estado conversacional fica em repositórios externos. |
| Segurança | Segredos são fornecidos por secret store ou Kubernetes Secrets. |
| Observabilidade | Logs, métricas e traces usam correlação por request_id, trace_id, session_id, tenant_id e agent_id. |
| Auditabilidade | Decisões de rota, guardrail, judge, MCP e LLM são rastreáveis. |
| Portabilidade | Execução suportada em local, Docker Compose e Kubernetes/OKE. |
| Configuração | Comportamento variável é controlado por `.env` e YAML versionado. |
### Critérios de Aceite
- [ ] Runtime recebe GatewayRequest validado.
- [ ] State contém tenant_id, agent_id, session_id, conversation_key, route e intent.
- [ ] Input guardrails executam antes do roteamento.
- [ ] Router ou Supervisor seleciona rota.
- [ ] Agent Node executa sem acessar payload bruto de canal.
- [ ] MCP é acessado por contrato.
- [ ] RAG é acessado por serviço reutilizável.
- [ ] Output guardrails executam antes da resposta final.
- [ ] Judges geram JudgeResult.
- [ ] Memória e checkpoint são persistidos conforme provider.
- [ ] Hosts transacionais declaram `active_transaction` e `last_transaction` no `AgentState`.
- [ ] Durante `COLLECTING_PARAMETERS`, respostas a parâmetros pendentes têm precedência sobre keyword routing genérico.
- [ ] Erros geram NOC e resposta controlada.
### Glossário
| Termo | Definição |
|---|---|
| Agent Platform | Plataforma composta por runtime, gateways, evaluator, templates, contratos e componentes operacionais. |
| Agent Framework | Biblioteca/core reutilizável com contratos, guardrails, judges, memória, telemetria, providers e utilitários. |
| Agent Runtime | Motor de execução de agentes baseado em LangGraph, estado, sessão, memória, checkpoints, roteamento e ciclo de vida. |
| Agent Gateway | Aplicação deployável de entrada, roteamento e orquestração entre backends/agentes. |
| Channel Gateway | Aplicação ou módulo de normalização de payloads de canais para GatewayRequest. |
| AI Gateway | Aplicação de governança, roteamento e abstração de chamadas LLM/embedding. |
| MCP Gateway | Aplicação de governança e roteamento de tools MCP. |
| Evaluator | Camada de avaliação online/offline, regressão e certificação. |
| Business Context | Conjunto de chaves canônicas de negócio: customer_key, contract_key, interaction_key, account_key, resource_key e session_key. |
### Operational performance and SRE requirements
> Consolidated from `specs/SPEC-020-Operational-Readiness-and-SRE-Model.md`.
### Agent Platform OCI
Version: 1.0.0
---
### Padrão de leitura
Cada SPEC está organizada para servir tanto como contrato arquitetural quanto como guia prático de adoção.
A estrutura usada é:
1. Conceito.
2. Problema que resolve.
3. Quando usar.
4. Quando não usar.
5. Arquitetura.
6. Implementação.
7. Exemplos.
8. Erros comuns.
9. Critérios de aceite.
---
### 1. Conceito
Operational Readiness define os requisitos mínimos para operar a Agent Platform OCI em produção com confiabilidade, observabilidade, capacidade de resposta a incidentes e recuperação.
### 2. Componentes operados
- Agent Gateway;
- Channel Gateway;
- Agent Runtime;
- AI Gateway;
- MCP Gateway;
- MCP Servers;
- Evaluator;
- bancos/repositórios;
- Langfuse/OTEL;
- Redis/Mongo/ADB quando usados.
### 3. Health e readiness
Endpoints mínimos:
```text
GET /health
GET /ready
GET /version
```
### 4. SLOs
| Componente | Latência | Disponibilidade |
| --- | --- | --- |
| Agent Gateway | p95 < 1s | 99.5% |
| Agent Runtime | p95 < 5s | 99.0% |
| AI Gateway | p95 < 10s | 99.0% |
| MCP Gateway | p95 < 2s | 99.0% |
| Evaluator | janela batch | execução diária |
### 5. Métricas
- requests_total;
- request_latency_ms;
- errors_total;
- active_sessions;
- llm_tokens_total;
- llm_cost_estimated;
- mcp_tool_calls_total;
- guardrail_blocks_total;
- judge_scores;
- evaluator_scores.
### 6. Dashboards
Dashboards mínimos:
- Platform Overview;
- Runtime;
- Gateway;
- AI Gateway;
- MCP Gateway;
- Guardrails;
- Evaluator;
- Cost/Usage;
- Incidents.
### 7. Alertas
| Alerta | Condição |
| --- | --- |
| HighErrorRate | 5xx acima do limite. |
| LatencySLOBreach | p95 acima do SLO. |
| LLMProviderDown | Falhas consecutivas no provider. |
| MCPTimeoutSpike | Aumento de timeout MCP. |
| GuardrailSpike | Aumento anômalo de bloqueios. |
| EvaluatorFailed | Run falhou. |
### 8. Runbooks
Runbook deve conter:
- sintoma;
- impacto;
- consultas;
- dashboards;
- logs;
- ações;
- rollback;
- escalonamento.
### 9. Incident management
Fluxo:
```mermaid
flowchart LR
Detect[Detect] --> Triage[Triage]
Triage --> Mitigate[Mitigate]
Mitigate --> Recover[Recover]
Recover --> Postmortem[Postmortem]
```
### 10. Capacidade
Avaliar:
- QPS;
- sessões simultâneas;
- tokens/minuto;
- chamadas MCP/minuto;
- latência de provider;
- uso de memória;
- storage de checkpoints.
### 11. Erros comuns
| Erro | Impacto | Correção |
| --- | --- | --- |
| Sem readiness | Tráfego antes do app estar pronto. | Implementar /ready. |
| Sem alertas MCP | Falha silenciosa. | Criar alertas por tool. |
| Sem runbook | MTTR alto. | Criar runbooks por incidente. |
| Sem custo LLM | Sem controle financeiro. | Registrar tokens/custos. |
### 12. Production readiness checklist
- [ ] Health checks ativos.
- [ ] Readiness checks ativos.
- [ ] Logs estruturados.
- [ ] Métricas exportadas.
- [ ] Traces exportados.
- [ ] Dashboards criados.
- [ ] Alertas configurados.
- [ ] Runbooks disponíveis.
- [ ] Rollback validado.
- [ ] SLOs definidos.
- [ ] Capacidade estimada.
- [ ] Incident process definido.
New fixes or evolutions for this subject should update this consolidated document. Release notes may continue to exist as history, but they should not be required to understand or implement the feature.

View File

@@ -1,129 +1,133 @@
### Developer Index — Agent Framework OCI
### How to use this documentation
The documentation has three clear levels:
1. **Main tutorial:** [`README_en.md`](../../../README_en.md) — build, configure, run and test an agent end to end.
2. **Architecture:** [01 — Architecture and Concepts](./01_architecture_and_concepts.md) — components, boundaries and implementation placement.
3. **Specialized references:** manuals `02` through `11`deep implementation and troubleshooting by capability.
1. **Main tutorial:** [`README_en.md`](README_en.md) — creation, configuration, execution, and testing of an agent from start to finish.
2. **Architecture:** [01 — Architecture and Concepts](docs/developer/en/01_architecture_and_concepts.md) — components, responsibilities, and where to implement each concern.
3. **Specialized references:** manuals `02` through `11`in-depth implementation and troubleshooting by capability.
If you are creating a new agent, start with the main README.
If you are starting a new agent, begin with `README_en.md`.
If something is not working, use **Search by problem** below.
### Search by problem
| Problem / question | Usually involves | Go to |
| Problem / question | What is usually involved | Where to look |
|---|---|---|
| Framework selects the wrong agent/intent | routing, intents, thresholds, deterministic/LLM mode | [Routing and Stickiness](./02_routing_stickiness_and_intent_shift.md) |
| Agent stays stuck on the same subject | route stickiness, intent shift, handoff | [Routing and Stickiness](./02_routing_stickiness_and_intent_shift.md) |
| A parameter answer is mistaken for a new intent | transaction precedence, parameter extraction | [Transactional Workflows](./03_transaction_workflows_and_state.md) |
| Transaction keeps asking for the same parameter | transaction state, extractor, schema | [Transactional Workflows](./03_transaction_workflows_and_state.md) and [MCP/Tools](./04_mcp_integration_tools_and_policies.md) |
| “yes/no” confirmation does not continue the flow | confirmation state | [Transactional Workflows](./03_transaction_workflows_and_state.md) |
| A closed transaction reappears | old checkpoint vs active transaction | [Transactional Workflows](./03_transaction_workflows_and_state.md) and [LTM/Checkpoint](./08_long_term_memory_and_checkpoint.md) |
| System claims an operation ran but there is no evidence | MCP results, `COMPLETED`, transaction judges | [Transactional Workflows](./03_transaction_workflows_and_state.md) and [Guardrails/Judges](./06_guardrails_judges_and_transaction_evaluation.md) |
| A tool is missing | tools config, MCP catalog/discovery | [MCP/Tools](./04_mcp_integration_tools_and_policies.md) |
| MCP Server is missing from catalog | registration, manifest/discovery, MCP Gateway | [MCP/Tools](./04_mcp_integration_tools_and_policies.md) and [Gateways](./05_agent_gateway_mcp_gateway_and_auth.md) |
| Tool parameters are wrong | schema, mapping, BusinessContext, extraction | [MCP/Tools](./04_mcp_integration_tools_and_policies.md) |
| Transactional tool executes without confirmation | policy, `require_confirmation` | [MCP/Tools](./04_mcp_integration_tools_and_policies.md) |
| 401 between gateway/backend/MCP | Basic Auth, hop credentials | [Gateways and Auth](./05_agent_gateway_mcp_gateway_and_auth.md) |
| Need to decide framework vs agent ownership | core/agent boundary | [Architecture and Concepts](./01_architecture_and_concepts.md) |
| Agent-specific guardrail breaks another agent | extension model, domain imports | [Guardrails and Judges](./06_guardrails_judges_and_transaction_evaluation.md) |
| Judge does not run for a transaction | sampling, transaction signals | [Guardrails and Judges](./06_guardrails_judges_and_transaction_evaluation.md) |
| Groundedness gets the wrong context | RAG context, MCP evidence, judge inputs | [RAG/Grounding](./07_rag_business_context_and_grounding.md) |
| RAG returns no useful content | provider, ingestion, embeddings | [RAG/Grounding](./07_rag_business_context_and_grounding.md) |
| Unsure whether to use RAG, memory or a tool | responsibility separation | [Architecture and Concepts](./01_architecture_and_concepts.md) |
| Memory disappears across sessions | LTM vs conversation memory | [LTM and Checkpoint](./08_long_term_memory_and_checkpoint.md) |
| Memory leaks across customer/agent | identity isolation | [LTM and Checkpoint](./08_long_term_memory_and_checkpoint.md) |
| Need `reasoning_content` | `ainvoke_response()` | [LLM Rich Response](./09_llm_rich_response_reasoning.md) |
| `reasoning_content` is `None` | provider/model does not expose it | [LLM Rich Response](./09_llm_rich_response_reasoning.md) |
| Too many LLM calls | deterministic routing, concurrency, cache | [Performance](./10_performance_cache_and_async_runtime.md) |
| Deadlock across event loops | cross-loop runtime/sequence | [Performance](./10_performance_cache_and_async_runtime.md) |
| Logs/traces do not correlate the same agent | labels, IDs, observability mapping | [Observability](./11_observability_persistence_and_operational_readiness.md) |
| Historical example no longer compiles | stale docs vs current API | [README Alignment Validation](./VALIDATION_README_ALIGNMENT.md) |
| Need to create a new agent from scratch | complete flow | [`README_en.md`](../../../README_en.md) |
| The framework does not find the correct agent/intent | routing, intents, threshold, deterministic/LLM mode | [Routing and Stickiness](docs/developer/en/02_routing_stickiness_and_intent_shift.md) |
| The agent gets stuck on the same subject and does not change intent | route stickiness, intent shift, handoff | [Routing and Stickiness](docs/developer/en/02_routing_stickiness_and_intent_shift.md) |
| An answer that should fill a parameter is interpreted as a new intent | transactional precedence, parameter extraction | [Transactional Workflows](docs/developer/en/03_transaction_workflows_and_state.md) |
| The transaction keeps asking for the same parameter | transaction state, extractor, schema | [Transactional Workflows](docs/developer/en/03_transaction_workflows_and_state.md) and [MCP/Tools](docs/developer/en/04_mcp_integration_tools_and_policies.md) |
| “yes/no” confirmation does not continue the flow | confirmation state, transaction state | [Transactional Workflows](docs/developer/en/03_transaction_workflows_and_state.md) |
| A completed transaction reappears | old checkpoint versus active transaction state | [Transactional Workflows](docs/developer/en/03_transaction_workflows_and_state.md) and [LTM/Checkpoint](docs/developer/en/08_long_term_memory_and_checkpoint.md) |
| The system says it executed something, but there is no evidence | MCP result, `COMPLETED` state, transactional judges | [Transactional Workflows](docs/developer/en/03_transaction_workflows_and_state.md) and [Guardrails/Judges](docs/developer/en/06_guardrails_judges_and_transaction_evaluation.md) |
| A tool does not appear or cannot be found | `tools.yaml`, MCP catalog, discovery | [MCP/Tools](docs/developer/en/04_mcp_integration_tools_and_policies.md) |
| MCP Server does not appear in the catalog | registration, manifest/discovery, MCP Gateway | [MCP/Tools](docs/developer/en/04_mcp_integration_tools_and_policies.md) and [Gateways](docs/developer/en/05_agent_gateway_mcp_gateway_and_auth.md) |
| Parameters sent to the tool are wrong | schema, mapping, BusinessContext, extractor | [MCP/Tools](docs/developer/en/04_mcp_integration_tools_and_policies.md) |
| A transactional operation executes without confirmation | tool policy, `require_confirmation` | [MCP/Tools](docs/developer/en/04_mcp_integration_tools_and_policies.md) |
| A name search requires an overly exact match | parameter extraction/mapping and agent logic | [MCP/Tools](docs/developer/en/04_mcp_integration_tools_and_policies.md) |
| I receive 401 between gateway/backend/MCP | Basic Auth, credentials per hop | [Gateways and Auth](docs/developer/en/05_agent_gateway_mcp_gateway_and_auth.md) |
| I need to decide whether something belongs to the framework or the agent | core/agent boundary | [Architecture and Concepts](docs/developer/en/01_architecture_and_concepts.md) |
| An agent-specific guardrail is breaking another agent | extensibility, domain imports in the core | [Guardrails and Judges](docs/developer/en/06_guardrails_judges_and_transaction_evaluation.md) |
| A judge does not run in a transaction | sampling, `always_run_for_transactional`, transaction signals | [Guardrails and Judges](docs/developer/en/06_guardrails_judges_and_transaction_evaluation.md) |
| Groundedness is evaluating without the correct context | RAG context, MCP evidence, judge inputs | [RAG/Grounding](docs/developer/en/07_rag_business_context_and_grounding.md) |
| RAG does not find content | provider, ingestion, embeddings, configuration | [RAG/Grounding](docs/developer/en/07_rag_business_context_and_grounding.md) |
| I do not know whether to use RAG, memory, or a tool | separation of responsibilities | [Architecture and Concepts](docs/developer/en/01_architecture_and_concepts.md) and [RAG/Grounding](docs/developer/en/07_rag_business_context_and_grounding.md) |
| Memory disappears when changing sessions | LTM versus conversation memory | [LTM and Checkpoint](docs/developer/en/08_long_term_memory_and_checkpoint.md) |
| Memory from one customer/agent appears in another | identity key, tenant/agent/customer isolation | [LTM and Checkpoint](docs/developer/en/08_long_term_memory_and_checkpoint.md) |
| I need to retrieve `reasoning_content` | `ainvoke_response()` | [LLM Rich Response](docs/developer/en/09_llm_rich_response_reasoning.md) |
| `reasoning_content` is `None` | provider/model does not expose the field | [LLM Rich Response](docs/developer/en/09_llm_rich_response_reasoning.md) |
| There are unnecessary LLM calls | deterministic routing, concurrency, cache | [Performance](docs/developer/en/10_performance_cache_and_async_runtime.md) |
| There is a deadlock or wait across event loops | cross-loop sequence/runtime | [Performance](docs/developer/en/10_performance_cache_and_async_runtime.md) |
| Logs/traces do not correlate the same agent | labels, IDs, and observability mapping | [Observability](docs/developer/en/11_observability_persistence_and_operational_readiness.md) |
| Sequence is interfering with processing | asynchronous sequence implementation | [Observability](docs/developer/en/11_observability_persistence_and_operational_readiness.md) and [Performance](docs/developer/en/10_performance_cache_and_async_runtime.md) |
| An old example does not compile | historical documentation versus current API | [README vs Code Validation](docs/developer/en/VALIDATION_README_ALIGNMENT.md) |
| I need to create a new agent from scratch | complete flow | [`README_en.md`](README_en.md) |
| I need to know where to place a new feature | architecture and boundaries | [Architecture and Concepts](docs/developer/en/01_architecture_and_concepts.md) |
### Search by feature
### [01 — Architecture and Concepts](./01_architecture_and_concepts.md)
### [01 — Architecture and Concepts](docs/developer/en/01_architecture_and_concepts.md)
**What it is:** component, contract and responsibility-boundary reference.
**What it is:** overview of components, contracts, and responsibility boundaries.
**Use it when:** understanding the platform or deciding where a feature belongs.
**Use when:** you need to understand the platform, decide where to implement something, or avoid coupling between core and agent.
### [02 — Routing, Route Stickiness and Intent Shift](./02_routing_stickiness_and_intent_shift.md)
### [02 — Routing, Route Stickiness, and Intent Shift](docs/developer/en/02_routing_stickiness_and_intent_shift.md)
**What it is:** agent/intent discovery, stickiness, handoff and intent-shift reference.
**What it is:** complete reference for agent/intent discovery, stickiness, handoff, and intent changes.
**Use it when:** routing is wrong or session continuity behaves incorrectly.
**Use when:** the message goes to the wrong agent, does not change intent, or loses continuity.
### [03 — Transactional Workflows and State](./03_transaction_workflows_and_state.md)
### [03 — Transactional Workflows and State](docs/developer/en/03_transaction_workflows_and_state.md)
**What it is:** multi-turn transaction lifecycle, states, confirmation, resume and execution evidence.
**What it is:** multi-turn transaction lifecycle, states, confirmation, pause/resume, and operational evidence.
**Use it when:** transactions loop, resume incorrectly or perform critical operations.
**Use when:** there are loops, incorrect confirmations, incorrect resumes, or critical operations.
### [04 — MCP, Tools, Policies and Parameter Extraction](./04_mcp_integration_tools_and_policies.md)
### [04 — MCP, Tools, Policies, and Parameter Extraction](docs/developer/en/04_mcp_integration_tools_and_policies.md)
**What it is:** tools, MCP Servers, mappings, policies and extraction reference.
**What it is:** reference for tools, MCP Servers, mappings, policies, and parameter extraction.
**Use it when:** building or troubleshooting tool integration.
**Use when:** tool integration/execution is incorrect or needs to be created.
### [05 — Agent Gateway, MCP Gateway and Authentication](./05_agent_gateway_mcp_gateway_and_auth.md)
### [05 — Agent Gateway, MCP Gateway, and Authentication](docs/developer/en/05_agent_gateway_mcp_gateway_and_auth.md)
**What it is:** gateway responsibilities, governance and component authentication.
**What it is:** gateway responsibilities, governance, and authentication between components.
**Use it when:** troubleshooting ingress, catalog, authorization or gateway deployment.
**Use when:** there is an ingress, catalog, authorization, 401, or gateway deployment problem.
### [06 — Guardrails, Judges and Transaction Evaluation](./06_guardrails_judges_and_transaction_evaluation.md)
### [06 — Guardrails, Judges, and Transaction Evaluation](docs/developer/en/06_guardrails_judges_and_transaction_evaluation.md)
**What it is:** native/external validation, judges, grounding and transaction evaluation.
**What it is:** native/external validations, judges, grounding, and rules for transactional turns.
**Use it when:** validation blocks, skips or evaluates incorrectly.
**Use when:** a validation blocks, does not run, or produces an incorrect evaluation.
### [07 — RAG, BusinessContext and Grounding](./07_rag_business_context_and_grounding.md)
### [07 — RAG, BusinessContext, and Grounding](docs/developer/en/07_rag_business_context_and_grounding.md)
**What it is:** RAG providers, retrieved context, BusinessContext and grounding.
**What it is:** RAG providers, retrieved context, BusinessContext, and grounding.
**Use it when:** retrieved knowledge does not reach the runtime/judge correctly.
**Use when:** retrieved knowledge does not correctly reach the agent/judge.
### [08 — Long-Term Memory and Checkpoint](./08_long_term_memory_and_checkpoint.md)
### [08 — Long-Term Memory and Checkpoint](docs/developer/en/08_long_term_memory_and_checkpoint.md)
**What it is:** durable memory, conversational memory, identity and state snapshots.
**What it is:** durable memory, conversation memory, identity, and state snapshots.
**Use it when:** context disappears, leaks or resumes incorrectly.
**Use when:** context disappears, leaks, or the workflow resumes from the wrong place.
### [09 — LLM Rich Response and reasoning_content](./09_llm_rich_response_reasoning.md)
### [09 — LLM Rich Response and reasoning_content](docs/developer/en/09_llm_rich_response_reasoning.md)
**What it is:** structured inference output beyond the `str` returned by `ainvoke()`.
**What it is:** structured inference response beyond the `str` returned by `ainvoke()`.
**Use it when:** consumers require provider metadata, usage or reasoning exposed by the provider.
**Use when:** consumers need metadata, usage, or reasoning exposed by the provider.
### [10 — Performance, Cache and Async Runtime](./10_performance_cache_and_async_runtime.md)
### [10 — Performance, Cache, and Async Runtime](docs/developer/en/10_performance_cache_and_async_runtime.md)
**What it is:** concurrency, caching, LLM and event-loop optimization reference.
**What it is:** concurrency, cache, LLM, and event-loop optimizations.
**Use it when:** reducing avoidable latency or diagnosing deadlocks.
**Use when:** there is avoidable latency, serial processing, or deadlock.
### [11 — Observability, Persistence and Operational Readiness](./11_observability_persistence_and_operational_readiness.md)
### [11 — Observability, Persistence, and Operational Readiness](docs/developer/en/11_observability_persistence_and_operational_readiness.md)
**What it is:** correlation, events, labels, sequencing, persistence and production diagnostics.
**What it is:** correlation, events, labels, sequence, persistence, and diagnostics.
**Use it when:** proving execution paths or diagnosing production behavior.
**Use when:** it is necessary to prove the executed path or diagnose production.
### Main tutorial
[`README_en.md`](../../../README_en.md) remains the complete step-by-step guide.
[`README_en.md`](README_en.md) remains the reference for the complete step-by-step flow:
`architecture → configuration → agent creation → registration → state → routing → tools → MCP → identity → execution → tests → gateways → memory → RAG`.
### Maintenance
Do not create another tutorial parallel to the root README.
Do not create another tutorial in parallel with `README_en.md`.
When a feature evolves:
When evolving a feature:
- update the README only when the normal developer flow changes;
- update the specialized manual with behavior, configuration, examples and troubleshooting;
- update SPECs when contracts change;
- update the README only if the normal development flow changed;
- update the specialized manual with behavior, configuration, examples, and troubleshooting;
- update SPECs if the contract changed;
- keep release notes as history, not as the only current documentation.

View File

@@ -1,27 +1,39 @@
### Documentation Alignment Validation
### Purpose
### Goal
Record how this version's documentation was reorganized and which sources developers should trust.
Record how the documentation for this version was reorganized and which sources developers should use.
### Structural decision
The root `README_en.md` / `README.md` is the **single end-to-end main tutorial**.
The root `README_en.md` is the **single end-to-end main tutorial**.
The former `01_architecture_and_agent_development.md` was removed because it repeated much of the README but not all of it. That created ambiguity: two documents appeared to teach the same workflow while one was partial.
The former `01_architecture_and_agent_development.md` was removed because it repeated a large part of the README, but not all of it. This created ambiguity: two documents appeared to teach the same thing, but one was partial.
The new structure replaces it with `01_architecture_and_concepts.md`, containing only architecture, concepts, responsibilities and extension criteria.
The new structure replaces that file with `01_architecture_and_concepts.md`, which contains only architecture, concepts, responsibilities, and extension criteria.
### `README_old2.md` validation
### Validation of `README_old2.md`
`Documentacao/README_old2.md` remains useful as historical material but is not the primary development source.
`Documentacao/README_old2.md` remains useful as history, but it is not the primary source for development.
Later evolution found in the current README/code includes SPECs/SDDs, richer `llm_profiles.yaml` guidance, Channel Gateway, canonical contracts, current memory composition, `RuntimeContext`, tool helpers, transaction helpers, direct MCP responses and gateway/RAG/memory/policy evolution.
Later evolutions were found in the current README and code, including:
### Main README correction
- SPECs/SDDs;
- more complete `llm_profiles.yaml` configuration;
- Channel Gateway and canonical contracts;
- `memory` and `summary_memory` in the current agent lifecycle;
- `prepare_memory_context()` and `build_messages()`;
- `RuntimeContext`;
- `normalize_tools_by_intent()`;
- `build_tool_arguments()`;
- `execute_tools_for_intent()`;
- transaction-state helpers;
- direct MCP responses;
- evolution of gateways, RAG, memory, and policies.
The generated package corrects this typo:
### Correction applied to the main README
The following typo was corrected in the generated package:
```python
from app.agents.financeiro_agent import FinanceirotAgent
@@ -33,7 +45,7 @@ to:
from app.agents.financeiro_agent import FinanceiroAgent
```
The correct class is confirmed by code and the rest of the documentation.
The correct class is confirmed by the code and the rest of the documentation.
### APIs confirmed in the current implementation
@@ -52,7 +64,7 @@ AgentRuntimeMixin.build_direct_mcp_answer()
### Trust order
1. version code;
1. code for the version;
2. main README for the same version;
3. SPECs/SDDs;
4. specialized manuals;
@@ -63,9 +75,9 @@ AgentRuntimeMixin.build_direct_mcp_answer()
A feature evolution should update:
1. the main README **only when the normal development path changes**;
2. the feature's specialized manual with technical detail, behavior, configuration and troubleshooting;
3. the SPEC when a contract changes;
4. a release note when historical recording is needed.
1. the main README, **only if it changes the normal development path**;
2. the feature's specialized manual, with technical details, behavior, configuration, and troubleshooting;
3. the SPEC, when there is a contract change;
4. the release note, when it is necessary to record the historical change.
Do not create another “main manual” for a feature. Do not keep functional corrections permanently only in release notes.
Do not create a new “main manual” for a feature. Do not keep functional fixes permanently only in release notes.

View File

@@ -728,6 +728,7 @@ GET /debug/env
### Arquitetura — Global Supervisor
```text
Usuário / Frontend
@@ -745,6 +746,8 @@ Usuário / Frontend
▼ ▼ ▼ ▼
Backend Backend Backend Backend
Contas Ofertas Suporte Cobrança
```
Cada backend continua sendo um projeto independente, com seus próprios agentes, prompts, MCPs e deploy, mas todos usam a mesma biblioteca agent_framework.
### Estado global
@@ -780,18 +783,21 @@ A capacidade usa um perfil LLM leve para decidir o tratamento global do turno se
### Fluxo
```text
Mensagem -> Classificador LLM leve
CONTINUE + agente ativo -> agente atual
ROUTE / baixa confiança / erro -> Enterprise Router
HUMAN_HANDOFF -> nó human_handoff
END_SESSION -> nó end_session
As ações globais podem ser reconhecidas no primeiro turno. Isso permite que “quero falar com uma pessoa” ou “pode encerrar” não dependam de um agente de domínio já selecionado.
```
### Configuração
### .env
```text
ENABLE_ROUTE_STICKINESS=true
ROUTE_STICKINESS_LLM_PROFILE=route_continuity
ROUTE_STICKINESS_CONFIDENCE_THRESHOLD=0.90
@@ -799,6 +805,7 @@ ROUTE_STICKINESS_HISTORY_TURNS=2
ROUTE_STICKINESS_MAX_TOKENS=80
HUMAN_HANDOFF_MESSAGE=Vou encaminhar seu atendimento para uma pessoa.
END_SESSION_MESSAGE=Atendimento encerrado. Obrigado pelo contato.
```
### llm_profiles.yaml