diff --git a/public/images/docs/agent-playground/node-connection-handles.png b/public/images/docs/agent-playground/node-connection-handles.png
new file mode 100644
index 00000000..a641ece1
Binary files /dev/null and b/public/images/docs/agent-playground/node-connection-handles.png differ
diff --git a/public/images/docs/knowledge-base/guides/create-knowledge-base.mp4 b/public/images/docs/knowledge-base/guides/create-knowledge-base.mp4
new file mode 100644
index 00000000..12d59710
Binary files /dev/null and b/public/images/docs/knowledge-base/guides/create-knowledge-base.mp4 differ
diff --git a/public/images/docs/knowledge-base/guides/update-knowledge-base.mp4 b/public/images/docs/knowledge-base/guides/update-knowledge-base.mp4
new file mode 100644
index 00000000..c25ac221
Binary files /dev/null and b/public/images/docs/knowledge-base/guides/update-knowledge-base.mp4 differ
diff --git a/public/images/docs/simulation/agent-development-loop.svg b/public/images/docs/simulation/agent-development-loop.svg
new file mode 100644
index 00000000..c48ebdf2
--- /dev/null
+++ b/public/images/docs/simulation/agent-development-loop.svg
@@ -0,0 +1,58 @@
+
diff --git a/public/images/docs/simulation/guides/connect-your-agent/chat-configuration.png b/public/images/docs/simulation/guides/connect-your-agent/chat-configuration.png
new file mode 100644
index 00000000..ae493e7f
Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/chat-configuration.png differ
diff --git a/public/images/docs/simulation/guides/connect-your-agent/connect-1.png b/public/images/docs/simulation/guides/connect-your-agent/connect-1.png
new file mode 100644
index 00000000..8b4c6a59
Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/connect-1.png differ
diff --git a/public/images/docs/simulation/guides/connect-your-agent/connect-2.png b/public/images/docs/simulation/guides/connect-your-agent/connect-2.png
new file mode 100644
index 00000000..b3d1c58c
Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/connect-2.png differ
diff --git a/public/images/docs/simulation/guides/connect-your-agent/connect-3.png b/public/images/docs/simulation/guides/connect-your-agent/connect-3.png
new file mode 100644
index 00000000..a674eb8b
Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/connect-3.png differ
diff --git a/public/images/docs/simulation/guides/connect-your-agent/connect-4.png b/public/images/docs/simulation/guides/connect-your-agent/connect-4.png
new file mode 100644
index 00000000..bfd43e0f
Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/connect-4.png differ
diff --git a/public/images/docs/simulation/guides/connect-your-agent/connect-5.png b/public/images/docs/simulation/guides/connect-your-agent/connect-5.png
new file mode 100644
index 00000000..48a9c1d0
Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/connect-5.png differ
diff --git a/public/images/docs/simulation/guides/connect-your-agent/create-new-version.mp4 b/public/images/docs/simulation/guides/connect-your-agent/create-new-version.mp4
new file mode 100644
index 00000000..bce36851
Binary files /dev/null and b/public/images/docs/simulation/guides/connect-your-agent/create-new-version.mp4 differ
diff --git a/public/images/docs/simulation/guides/create-personas/chat-settings.png b/public/images/docs/simulation/guides/create-personas/chat-settings.png
new file mode 100644
index 00000000..68497719
Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/chat-settings.png differ
diff --git a/public/images/docs/simulation/guides/create-personas/choose-persona-type.png b/public/images/docs/simulation/guides/create-personas/choose-persona-type.png
new file mode 100644
index 00000000..8bf979c9
Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/choose-persona-type.png differ
diff --git a/public/images/docs/simulation/guides/create-personas/conversation-settings.png b/public/images/docs/simulation/guides/create-personas/conversation-settings.png
new file mode 100644
index 00000000..2ed07e8a
Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/conversation-settings.png differ
diff --git a/public/images/docs/simulation/guides/create-personas/custom-properties.png b/public/images/docs/simulation/guides/create-personas/custom-properties.png
new file mode 100644
index 00000000..73d19bef
Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/custom-properties.png differ
diff --git a/public/images/docs/simulation/guides/create-personas/custom-tab.png b/public/images/docs/simulation/guides/create-personas/custom-tab.png
new file mode 100644
index 00000000..e89c55de
Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/custom-tab.png differ
diff --git a/public/images/docs/simulation/guides/create-personas/open-personas.png b/public/images/docs/simulation/guides/create-personas/open-personas.png
new file mode 100644
index 00000000..8fd9012b
Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/open-personas.png differ
diff --git a/public/images/docs/simulation/guides/create-personas/persona-details.png b/public/images/docs/simulation/guides/create-personas/persona-details.png
new file mode 100644
index 00000000..6c5b8e57
Binary files /dev/null and b/public/images/docs/simulation/guides/create-personas/persona-details.png differ
diff --git a/public/images/docs/simulation/guides/create-scenarios/add-personas-by-default.png b/public/images/docs/simulation/guides/create-scenarios/add-personas-by-default.png
new file mode 100644
index 00000000..4686637c
Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/add-personas-by-default.png differ
diff --git a/public/images/docs/simulation/guides/create-scenarios/agent-definition-toggle.png b/public/images/docs/simulation/guides/create-scenarios/agent-definition-toggle.png
new file mode 100644
index 00000000..81f1107f
Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/agent-definition-toggle.png differ
diff --git a/public/images/docs/simulation/guides/create-scenarios/basic-information.png b/public/images/docs/simulation/guides/create-scenarios/basic-information.png
new file mode 100644
index 00000000..2d648db9
Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/basic-information.png differ
diff --git a/public/images/docs/simulation/guides/create-scenarios/call-chat-sop.png b/public/images/docs/simulation/guides/create-scenarios/call-chat-sop.png
new file mode 100644
index 00000000..2354041e
Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/call-chat-sop.png differ
diff --git a/public/images/docs/simulation/guides/create-scenarios/import-datasets.png b/public/images/docs/simulation/guides/create-scenarios/import-datasets.png
new file mode 100644
index 00000000..9582db74
Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/import-datasets.png differ
diff --git a/public/images/docs/simulation/guides/create-scenarios/open-scenarios.png b/public/images/docs/simulation/guides/create-scenarios/open-scenarios.png
new file mode 100644
index 00000000..53087530
Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/open-scenarios.png differ
diff --git a/public/images/docs/simulation/guides/create-scenarios/upload-script.png b/public/images/docs/simulation/guides/create-scenarios/upload-script.png
new file mode 100644
index 00000000..e265f2bf
Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/upload-script.png differ
diff --git a/public/images/docs/simulation/guides/create-scenarios/workflow-builder.png b/public/images/docs/simulation/guides/create-scenarios/workflow-builder.png
new file mode 100644
index 00000000..240fdf6c
Binary files /dev/null and b/public/images/docs/simulation/guides/create-scenarios/workflow-builder.png differ
diff --git a/public/images/docs/simulation/guides/create-simulation/add-simulation-details.png b/public/images/docs/simulation/guides/create-simulation/add-simulation-details.png
new file mode 100644
index 00000000..312154c5
Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/add-simulation-details.png differ
diff --git a/public/images/docs/simulation/guides/create-simulation/choose-scenarios.png b/public/images/docs/simulation/guides/create-simulation/choose-scenarios.png
new file mode 100644
index 00000000..b9d57fe6
Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/choose-scenarios.png differ
diff --git a/public/images/docs/simulation/guides/create-simulation/configure-eval.png b/public/images/docs/simulation/guides/create-simulation/configure-eval.png
new file mode 100644
index 00000000..896329d8
Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/configure-eval.png differ
diff --git a/public/images/docs/simulation/guides/create-simulation/eval-library.png b/public/images/docs/simulation/guides/create-simulation/eval-library.png
new file mode 100644
index 00000000..f1bd1927
Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/eval-library.png differ
diff --git a/public/images/docs/simulation/guides/create-simulation/open-run-simulation.png b/public/images/docs/simulation/guides/create-simulation/open-run-simulation.png
new file mode 100644
index 00000000..c74775a9
Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/open-run-simulation.png differ
diff --git a/public/images/docs/simulation/guides/create-simulation/sdk-boilerplate.png b/public/images/docs/simulation/guides/create-simulation/sdk-boilerplate.png
new file mode 100644
index 00000000..70dcc436
Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/sdk-boilerplate.png differ
diff --git a/public/images/docs/simulation/guides/create-simulation/select-evaluations.png b/public/images/docs/simulation/guides/create-simulation/select-evaluations.png
new file mode 100644
index 00000000..622abb5e
Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/select-evaluations.png differ
diff --git a/public/images/docs/simulation/guides/create-simulation/selected-evaluations.png b/public/images/docs/simulation/guides/create-simulation/selected-evaluations.png
new file mode 100644
index 00000000..bfc264cb
Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/selected-evaluations.png differ
diff --git a/public/images/docs/simulation/guides/create-simulation/summary.mp4 b/public/images/docs/simulation/guides/create-simulation/summary.mp4
new file mode 100644
index 00000000..6e253667
Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/summary.mp4 differ
diff --git a/public/images/docs/simulation/guides/create-simulation/voice-run-detail.png b/public/images/docs/simulation/guides/create-simulation/voice-run-detail.png
new file mode 100644
index 00000000..3bc30c2c
Binary files /dev/null and b/public/images/docs/simulation/guides/create-simulation/voice-run-detail.png differ
diff --git a/public/images/docs/simulation/guides/explore-scenarios/add-columns-define-column.png b/public/images/docs/simulation/guides/explore-scenarios/add-columns-define-column.png
new file mode 100644
index 00000000..a8e64825
Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/add-columns-define-column.png differ
diff --git a/public/images/docs/simulation/guides/explore-scenarios/add-columns-open-form.png b/public/images/docs/simulation/guides/explore-scenarios/add-columns-open-form.png
new file mode 100644
index 00000000..3370e57d
Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/add-columns-open-form.png differ
diff --git a/public/images/docs/simulation/guides/explore-scenarios/add-rows-choose-method.png b/public/images/docs/simulation/guides/explore-scenarios/add-rows-choose-method.png
new file mode 100644
index 00000000..243bab4e
Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/add-rows-choose-method.png differ
diff --git a/public/images/docs/simulation/guides/explore-scenarios/add-rows-describe-ai-rows.png b/public/images/docs/simulation/guides/explore-scenarios/add-rows-describe-ai-rows.png
new file mode 100644
index 00000000..5c0aac12
Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/add-rows-describe-ai-rows.png differ
diff --git a/public/images/docs/simulation/guides/explore-scenarios/add-rows-set-empty-row-count.png b/public/images/docs/simulation/guides/explore-scenarios/add-rows-set-empty-row-count.png
new file mode 100644
index 00000000..dfa4a20e
Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/add-rows-set-empty-row-count.png differ
diff --git a/public/images/docs/simulation/guides/explore-scenarios/graph-add-node.png b/public/images/docs/simulation/guides/explore-scenarios/graph-add-node.png
new file mode 100644
index 00000000..65150521
Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/graph-add-node.png differ
diff --git a/public/images/docs/simulation/guides/explore-scenarios/graph-click-node.png b/public/images/docs/simulation/guides/explore-scenarios/graph-click-node.png
new file mode 100644
index 00000000..18d35745
Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/graph-click-node.png differ
diff --git a/public/images/docs/simulation/guides/explore-scenarios/graph-node-details.png b/public/images/docs/simulation/guides/explore-scenarios/graph-node-details.png
new file mode 100644
index 00000000..20c7c441
Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/graph-node-details.png differ
diff --git a/public/images/docs/simulation/guides/explore-scenarios/graph-open-editor.png b/public/images/docs/simulation/guides/explore-scenarios/graph-open-editor.png
new file mode 100644
index 00000000..9684a48d
Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/graph-open-editor.png differ
diff --git a/public/images/docs/simulation/guides/explore-scenarios/overview-open-scenario.png b/public/images/docs/simulation/guides/explore-scenarios/overview-open-scenario.png
new file mode 100644
index 00000000..83924100
Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/overview-open-scenario.png differ
diff --git a/public/images/docs/simulation/guides/explore-scenarios/overview-scenario-detail.png b/public/images/docs/simulation/guides/explore-scenarios/overview-scenario-detail.png
new file mode 100644
index 00000000..a0c253fe
Binary files /dev/null and b/public/images/docs/simulation/guides/explore-scenarios/overview-scenario-detail.png differ
diff --git a/public/images/docs/simulation/replay-loop.svg b/public/images/docs/simulation/replay-loop.svg
new file mode 100644
index 00000000..8aed85a7
--- /dev/null
+++ b/public/images/docs/simulation/replay-loop.svg
@@ -0,0 +1,41 @@
+
diff --git a/public/images/docs/simulation/simulation-model-agent-highlighted.svg b/public/images/docs/simulation/simulation-model-agent-highlighted.svg
new file mode 100644
index 00000000..7f8e2b65
--- /dev/null
+++ b/public/images/docs/simulation/simulation-model-agent-highlighted.svg
@@ -0,0 +1,53 @@
+
diff --git a/public/images/docs/simulation/simulation-model-personas-highlighted.svg b/public/images/docs/simulation/simulation-model-personas-highlighted.svg
new file mode 100644
index 00000000..10a14de8
--- /dev/null
+++ b/public/images/docs/simulation/simulation-model-personas-highlighted.svg
@@ -0,0 +1,53 @@
+
diff --git a/public/images/docs/simulation/simulation-model-scenario-highlighted.svg b/public/images/docs/simulation/simulation-model-scenario-highlighted.svg
new file mode 100644
index 00000000..cbbbca61
--- /dev/null
+++ b/public/images/docs/simulation/simulation-model-scenario-highlighted.svg
@@ -0,0 +1,53 @@
+
diff --git a/public/images/docs/simulation/simulation-model.svg b/public/images/docs/simulation/simulation-model.svg
new file mode 100644
index 00000000..2936ceb4
--- /dev/null
+++ b/public/images/docs/simulation/simulation-model.svg
@@ -0,0 +1,53 @@
+
diff --git a/src/lib/navigation.ts b/src/lib/navigation.ts
index af21b375..ddc6ebff 100644
--- a/src/lib/navigation.ts
+++ b/src/lib/navigation.ts
@@ -107,11 +107,32 @@ export const tabNavigation: NavTab[] = [
]
},
{
- title: 'Features',
+ title: 'Guides',
items: [
- { title: 'Create a Graph', href: '/docs/agent-playground/features/create-graph' },
- { title: 'Build a Workflow', href: '/docs/agent-playground/features/build-workflow' },
- { title: 'Run & Monitor', href: '/docs/agent-playground/features/run-and-monitor' },
+ { title: 'Create an agent', href: '/docs/agent-playground/guides/create-agent' },
+ {
+ title: 'Build a workflow',
+ items: [
+ { title: 'Overview', href: '/docs/agent-playground/guides/build-workflow' },
+ { title: 'Configure an LLM Prompt node', href: '/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node' },
+ { title: 'Configure an Agent node', href: '/docs/agent-playground/guides/build-workflow/configure-an-agent-node' },
+ { title: 'Set input variables', href: '/docs/agent-playground/guides/build-workflow/set-input-variables' },
+ ]
+ },
+ { title: 'Run an agent', href: '/docs/agent-playground/guides/run-an-agent' },
+ { title: 'Manage versions', href: '/docs/agent-playground/guides/manage-versions' },
+ ]
+ },
+ {
+ title: 'Reference',
+ items: [
+ { title: 'Limits & rules', href: '/docs/agent-playground/reference/limits-and-rules' },
+ ]
+ },
+ {
+ title: 'Troubleshooting',
+ items: [
+ { title: 'Agent Playground FAQ & fixes', href: '/docs/agent-playground/troubleshooting' },
]
},
]
@@ -124,28 +145,44 @@ export const tabNavigation: NavTab[] = [
{
title: 'Concepts',
items: [
+ { title: 'Understanding Annotation', href: '/docs/annotations/concepts/understanding-annotation' },
+ { title: 'Labels', href: '/docs/annotations/concepts/labels' },
+ { title: 'Queues & Items', href: '/docs/annotations/concepts/queues-and-items' },
{ title: 'Scores', href: '/docs/annotations/concepts/scores' },
]
},
{
- title: 'Features',
+ title: 'Guides',
items: [
- { title: 'Labels', href: '/docs/annotations/features/labels' },
- { title: 'Queues', href: '/docs/annotations/features/queues' },
- { title: 'Add Items to Queues', href: '/docs/annotations/features/add-items' },
- { title: 'Annotate Items', href: '/docs/annotations/features/annotate' },
- { title: 'Inline Annotations', href: '/docs/annotations/features/inline' },
- { title: 'Analytics & Agreement', href: '/docs/annotations/features/analytics' },
- { title: 'Export Annotations', href: '/docs/annotations/features/export' },
- { title: 'Automation Rules', href: '/docs/annotations/features/automation' },
+ { title: 'Create a label', href: '/docs/annotations/guides/create-label' },
+ { title: 'Create a queue', href: '/docs/annotations/guides/create-queue' },
+ {
+ title: 'Explore a queue',
+ items: [
+ { title: 'Overview', href: '/docs/annotations/guides/explore-queue' },
+ { title: 'Add items', href: '/docs/annotations/guides/explore-queue/add-items' },
+ { title: 'Track progress & agreement', href: '/docs/annotations/guides/explore-queue/progress-and-agreement' },
+ { title: 'Automate item intake', href: '/docs/annotations/guides/explore-queue/automate-item-intake' },
+ ]
+ },
+ { title: 'Annotate items', href: '/docs/annotations/guides/annotate-items' },
+ { title: 'Review submissions', href: '/docs/annotations/guides/review-submissions' },
+ { title: 'Annotate without a queue', href: '/docs/annotations/guides/annotate-without-a-queue' },
+ { title: 'Export annotations', href: '/docs/annotations/guides/export-annotations' },
]
},
{
- title: 'SDK',
+ title: 'Reference',
items: [
- { title: 'Python SDK', href: '/docs/annotations/sdk/python' },
- { title: 'JavaScript SDK', href: '/docs/annotations/sdk/javascript' },
- { title: 'Annotation Queue Using SDK', href: '/docs/annotations/sdk/annotation-queue-using-sdk' },
+ { title: 'Label types & values', href: '/docs/annotations/reference/label-types-and-values' },
+ { title: 'Queue settings & limits', href: '/docs/annotations/reference/queue-settings-and-limits' },
+ { title: 'SDK & API', href: '/docs/annotations/reference/sdk-api' },
+ ]
+ },
+ {
+ title: 'Troubleshooting',
+ items: [
+ { title: 'Annotation FAQ & fixes', href: '/docs/annotations/troubleshooting' },
]
},
]
@@ -253,21 +290,33 @@ export const tabNavigation: NavTab[] = [
{
title: 'Concepts',
items: [
- { title: 'Understanding Datasets', href: '/docs/dataset/concept/understanding-dataset' },
- { title: 'Static Columns', href: '/docs/dataset/concept/static-column' },
- { title: 'Dynamic Columns', href: '/docs/dataset/concept/dynamic-column' },
- { title: 'Synthetic Data', href: '/docs/dataset/concept/synthetic-data' },
+ { title: 'Understanding Datasets', href: '/docs/dataset/concepts/understanding-datasets' },
+ { title: 'Static & Dynamic Columns', href: '/docs/dataset/concepts/static-and-dynamic-columns' },
+ { title: 'Synthetic Data', href: '/docs/dataset/concepts/synthetic-data' },
]
},
{
- title: 'Features',
+ title: 'Guides',
items: [
- { title: 'Create New Dataset', href: '/docs/dataset/features/create' },
- { title: 'Add Rows to Dataset', href: '/docs/dataset/features/add-rows' },
- { title: 'Add Columns to Dataset', href: '/docs/dataset/features/add-columns' },
- { title: 'Run Prompt in Dataset', href: '/docs/dataset/features/run-prompt' },
- { title: 'Experiments in Dataset', href: '/docs/dataset/features/experiments' },
- { title: 'Add Annotation', href: '/docs/dataset/features/annotate' },
+ { title: 'Create a dataset', href: '/docs/dataset/guides/create-a-dataset' },
+ { title: 'Add rows', href: '/docs/dataset/guides/add-rows' },
+ { title: 'Add columns', href: '/docs/dataset/guides/add-columns' },
+ { title: 'Run a prompt on every row', href: '/docs/dataset/guides/run-a-prompt-on-every-row' },
+ { title: 'Run an experiment', href: '/docs/dataset/guides/run-an-experiment' },
+ { title: 'Manage datasets', href: '/docs/dataset/guides/manage-datasets' },
+ ]
+ },
+ {
+ title: 'Reference',
+ items: [
+ { title: 'Limits & Data Types', href: '/docs/dataset/reference/limits-and-data-types' },
+ { title: 'Dynamic column methods', href: '/docs/dataset/reference/dynamic-column-methods' },
+ ]
+ },
+ {
+ title: 'Troubleshooting',
+ items: [
+ { title: 'Dataset FAQ & fixes', href: '/docs/dataset/troubleshooting' },
]
},
]
@@ -280,25 +329,34 @@ export const tabNavigation: NavTab[] = [
{
title: 'Concepts',
items: [
- { title: 'How It Works', href: '/docs/error-feed/concepts/how-it-works' },
- { title: 'Error Taxonomy', href: '/docs/error-feed/concepts/taxonomy' },
- { title: 'Scoring', href: '/docs/error-feed/concepts/scoring' },
- { title: 'Severity and Status', href: '/docs/error-feed/concepts/severity-and-status' },
+ { title: 'Understanding Error Feed', href: '/docs/error-feed/concepts/understanding-error-feed' },
+ { title: 'Severity & Status', href: '/docs/error-feed/concepts/severity-and-status' },
+ { title: 'Trace error analysis', href: '/docs/error-feed/concepts/trace-error-analysis' },
]
},
{
- title: 'Features',
+ title: 'Guides',
items: [
- { title: 'The Feed', href: '/docs/error-feed/features/the-feed' },
- { title: 'Issue Overview', href: '/docs/error-feed/features/issue-overview' },
- { title: 'Traces', href: '/docs/error-feed/features/traces' },
- { title: 'State Graph', href: '/docs/error-feed/features/state-graph' },
- { title: 'Trends', href: '/docs/error-feed/features/trends' },
- { title: 'Metadata Panel', href: '/docs/error-feed/features/metadata-panel' },
- { title: 'Triage Workflow', href: '/docs/error-feed/features/triage-workflow' },
- { title: 'Deep Analysis', href: '/docs/error-feed/features/deep-analysis' },
- { title: 'Linear Integration', href: '/docs/error-feed/features/linear-integration' },
- { title: 'Sampling', href: '/docs/error-feed/features/sampling' },
+ { title: 'Turn on Error Feed', href: '/docs/error-feed/guides/turn-on-error-feed' },
+ { title: 'Triage issues', href: '/docs/error-feed/guides/triage-issues' },
+ { title: 'Investigate an issue', href: '/docs/error-feed/guides/investigate-an-issue' },
+ { title: 'Run a root cause analysis', href: '/docs/error-feed/guides/run-root-cause-analysis' },
+ { title: 'Create a Linear issue', href: '/docs/error-feed/guides/create-linear-issue' },
+ ]
+ },
+ {
+ title: 'Reference',
+ items: [
+ { title: 'Issue fields & filters', href: '/docs/error-feed/reference/issue-fields' },
+ { title: 'Error taxonomy', href: '/docs/error-feed/reference/error-taxonomy' },
+ ]
+ },
+ {
+ title: 'Troubleshooting',
+ items: [
+ { title: 'No issues in the feed', href: '/docs/error-feed/troubleshooting/no-issues-in-the-feed' },
+ { title: 'Issue counts look wrong', href: '/docs/error-feed/troubleshooting/issue-counts-look-wrong' },
+ { title: "Analysis doesn't finish", href: '/docs/error-feed/troubleshooting/analysis-does-not-finish' },
]
},
]
@@ -501,11 +559,22 @@ export const tabNavigation: NavTab[] = [
items: [
{ title: 'Overview', href: '/docs/falcon-ai' },
{
- title: 'Features',
+ title: 'Concepts',
+ items: [
+ { title: 'Understanding Falcon AI', href: '/docs/falcon-ai/concepts/understanding-falcon-ai' },
+ { title: 'Skills', href: '/docs/falcon-ai/concepts/skills' },
+ { title: 'MCP Connectors', href: '/docs/falcon-ai/concepts/mcp-connectors' },
+ ]
+ },
+ {
+ title: 'Guides',
items: [
- { title: 'Using Falcon AI', href: '/docs/falcon-ai/features/chat' },
- { title: 'Skill Builder', href: '/docs/falcon-ai/features/skills' },
- { title: 'MCP Connectors', href: '/docs/falcon-ai/features/mcp-connectors' },
+ { title: 'Chat with Falcon AI', href: '/docs/falcon-ai/guides/chat-with-falcon-ai' },
+ { title: 'Manage conversations', href: '/docs/falcon-ai/guides/manage-conversations' },
+ { title: 'Create a skill', href: '/docs/falcon-ai/guides/create-skill' },
+ { title: 'Connect an MCP server', href: '/docs/falcon-ai/guides/connect-mcp-server' },
+ { title: 'Choose connector tools', href: '/docs/falcon-ai/guides/choose-connector-tools' },
+ { title: 'Use the MCP Server in your IDE', href: '/docs/falcon-ai/guides/use-the-mcp-server' },
]
},
]
@@ -518,14 +587,15 @@ export const tabNavigation: NavTab[] = [
{
title: 'Concepts',
items: [
- { title: 'Understanding Knowledge Base', href: '/docs/knowledge-base/concepts/concept' },
+ { title: 'Understanding Knowledge Base', href: '/docs/knowledge-base/concepts/understanding-knowledge-base' },
]
},
{
- title: 'Features',
+ title: 'Guides',
items: [
- { title: 'Create KB Using SDK', href: '/docs/knowledge-base/features/sdk' },
- { title: 'Create KB Using UI', href: '/docs/knowledge-base/features/ui' },
+ { title: 'Create a knowledge base', href: '/docs/knowledge-base/guides/create-knowledge-base' },
+ { title: 'Update a knowledge base', href: '/docs/knowledge-base/guides/update-knowledge-base' },
+ { title: 'Manage with the SDK', href: '/docs/knowledge-base/guides/manage-with-the-sdk' },
]
},
]
@@ -589,20 +659,40 @@ export const tabNavigation: NavTab[] = [
{
title: 'Concepts',
items: [
- { title: 'Understanding Optimization', href: '/docs/optimization/concepts/concept' },
- { title: 'Bayesian Search', href: '/docs/optimization/optimizers/bayesian-search' },
- { title: 'Meta-Prompt', href: '/docs/optimization/optimizers/meta-prompt' },
- { title: 'ProTeGi', href: '/docs/optimization/optimizers/protegi' },
- { title: 'PromptWizard', href: '/docs/optimization/optimizers/promptwizard' },
- { title: 'GEPA', href: '/docs/optimization/optimizers/gepa' },
- { title: 'Random Search', href: '/docs/optimization/optimizers/random-search' },
+ { title: 'Understanding optimization', href: '/docs/optimization/concepts/understanding-optimization' },
+ { title: 'Choosing an optimizer', href: '/docs/optimization/concepts/choosing-an-optimizer' },
]
},
{
- title: 'Features',
+ title: 'Guides',
items: [
- { title: 'Using Python SDK', href: '/docs/optimization/features/using-python-sdk' },
- { title: 'Using Platform', href: '/docs/optimization/features/using-platform' },
+ { title: 'Run an optimization', href: '/docs/optimization/guides/run-an-optimization' },
+ { title: 'Read optimization results', href: '/docs/optimization/guides/read-optimization-results' },
+ { title: 'Optimize from the SDK', href: '/docs/optimization/guides/optimize-from-the-sdk' },
+ ]
+ },
+ {
+ title: 'Reference',
+ items: [
+ {
+ title: 'Optimizers',
+ items: [
+ { title: 'Overview', href: '/docs/optimization/reference/optimizers' },
+ { title: 'Random Search', href: '/docs/optimization/reference/optimizers/random-search' },
+ { title: 'Bayesian Search', href: '/docs/optimization/reference/optimizers/bayesian-search' },
+ { title: 'ProTeGi', href: '/docs/optimization/reference/optimizers/protegi' },
+ { title: 'Meta-Prompt', href: '/docs/optimization/reference/optimizers/meta-prompt' },
+ { title: 'PromptWizard', href: '/docs/optimization/reference/optimizers/promptwizard' },
+ { title: 'GEPA', href: '/docs/optimization/reference/optimizers/gepa' },
+ ]
+ },
+ { title: 'SDK & API', href: '/docs/optimization/reference/sdk-api' },
+ ]
+ },
+ {
+ title: 'Troubleshooting',
+ items: [
+ { title: 'Optimization FAQ & fixes', href: '/docs/optimization/troubleshooting' },
]
},
]
@@ -615,20 +705,33 @@ export const tabNavigation: NavTab[] = [
{
title: 'Concepts',
items: [
- { title: 'Prompt Engineering', href: '/docs/prompt/concepts/prompt-engineering' },
{ title: 'Understanding Prompts', href: '/docs/prompt/concepts/understanding-prompts' },
- { title: 'Versions and Labels', href: '/docs/prompt/concepts/versions-and-labels' },
+ { title: 'Versions & Labels', href: '/docs/prompt/concepts/versions-and-labels' },
+ { title: 'Prompt Engineering', href: '/docs/prompt/concepts/prompt-engineering' },
]
},
{
- title: 'Features',
+ title: 'Guides',
items: [
- { title: 'Create Prompt from Scratch', href: '/docs/prompt/features/create-from-scratch' },
- { title: 'Create from Existing Template', href: '/docs/prompt/features/create-from-template' },
- { title: 'Create with AI', href: '/docs/prompt/features/create-with-ai' },
- { title: 'Prompt Workbench Using SDK', href: '/docs/prompt/features/sdk' },
- { title: 'Linked Traces', href: '/docs/prompt/features/linked-traces' },
- { title: 'Manage Folders', href: '/docs/prompt/features/folders' },
+ { title: 'Create a prompt', href: '/docs/prompt/guides/create-a-prompt' },
+ { title: 'Run a prompt', href: '/docs/prompt/guides/run-a-prompt' },
+ { title: 'Commit & compare versions', href: '/docs/prompt/guides/commit-and-compare-versions' },
+ { title: 'Evaluate prompt outputs', href: '/docs/prompt/guides/evaluate-prompt-outputs' },
+ { title: 'Track prompt performance', href: '/docs/prompt/guides/track-prompt-performance' },
+ { title: 'Organize prompts in folders', href: '/docs/prompt/guides/organize-prompts-in-folders' },
+ ]
+ },
+ {
+ title: 'Reference',
+ items: [
+ { title: 'Model configuration', href: '/docs/prompt/reference/model-configuration' },
+ { title: 'SDK & API', href: '/docs/prompt/reference/sdk-api' },
+ ]
+ },
+ {
+ title: 'Troubleshooting',
+ items: [
+ { title: 'Prompt FAQ & fixes', href: '/docs/prompt/troubleshooting' },
]
},
]
@@ -641,35 +744,30 @@ export const tabNavigation: NavTab[] = [
{
title: 'Concepts',
items: [
- { title: 'Use Cases', href: '/docs/protect/concepts/concept' },
+ { title: 'Understanding Protect', href: '/docs/protect/concepts/understanding-protect' },
]
},
{
- title: 'Features',
+ title: 'Guides',
items: [
- { title: 'Run Protect via SDK', href: '/docs/protect/features/run-protect' },
+ { title: 'Turn on a guardrail', href: '/docs/protect/guides/turn-on-a-guardrail' },
+ { title: 'Test a guardrail', href: '/docs/protect/guides/test-a-guardrail' },
+ { title: 'Review guardrail activity', href: '/docs/protect/guides/review-guardrail-activity' },
+ { title: 'Run Protect from the SDK', href: '/docs/protect/guides/run-protect-from-the-sdk' },
]
},
- ]
- },
- {
- group: 'Prototype',
- icon: 'flask',
- items: [
- { title: 'Overview', href: '/docs/prototype' },
{
- title: 'Concepts',
+ title: 'Reference',
items: [
- { title: 'Understanding Prototype', href: '/docs/prototype/concepts/understanding-prototype' },
- { title: 'Versions and Runs', href: '/docs/prototype/concepts/versions-and-runs' },
+ { title: 'Guardrail checks', href: '/docs/protect/reference/guardrail-checks' },
]
},
{
- title: 'Features',
+ title: 'Troubleshooting',
items: [
- { title: 'Set Up Prototype', href: '/docs/prototype/features/set-up-prototype' },
- { title: 'Evals', href: '/docs/prototype/features/evals' },
- { title: 'Choose Winner', href: '/docs/prototype/features/choose-winner' },
+ { title: 'Guardrail changes not taking effect', href: '/docs/protect/troubleshooting/guardrail-changes-not-taking-effect' },
+ { title: 'Guardrail fires on the wrong requests', href: '/docs/protect/troubleshooting/guardrail-fires-on-the-wrong-requests' },
+ { title: 'Protect SDK rejects an input', href: '/docs/protect/troubleshooting/protect-sdk-rejects-an-input' },
]
},
]
@@ -706,28 +804,84 @@ export const tabNavigation: NavTab[] = [
{
title: 'Concepts',
items: [
- { title: 'Agent Definition', href: '/docs/simulation/concepts/agent-definition' },
+ { title: 'Understanding Simulation', href: '/docs/simulation/concepts/understanding-simulation' },
+ { title: 'Agent definitions & versions', href: '/docs/simulation/concepts/agent-definitions' },
{ title: 'Scenarios', href: '/docs/simulation/concepts/scenarios' },
{ title: 'Personas', href: '/docs/simulation/concepts/personas' },
- { title: 'Global Nodes', href: '/docs/simulation/concepts/global-nodes' },
+ { title: 'Runs & results', href: '/docs/simulation/concepts/runs-and-results' },
+ { title: 'Replay', href: '/docs/simulation/concepts/replay' },
+ { title: 'Optimization', href: '/docs/simulation/concepts/optimization' },
]
},
{
- title: 'Features',
+ title: 'Guides',
items: [
- { title: 'Run Voice Simulation', href: '/docs/simulation/features/run-simulation' },
- { title: 'Chat Simulation Using SDK', href: '/docs/simulation/features/simulation-using-sdk' },
+ { title: 'Connect your agent', href: '/docs/simulation/guides/connect-your-agent' },
+ { title: 'Create scenarios', href: '/docs/simulation/guides/create-scenarios' },
+ { title: 'Create personas', href: '/docs/simulation/guides/create-personas' },
{
- title: 'Replay',
+ title: 'Explore scenarios',
items: [
- { title: 'Chat Replay', href: '/docs/simulation/features/observe-to-simulate' },
- { title: 'Voice Replay', href: '/docs/simulation/features/voice-replay' },
+ { title: 'Overview', href: '/docs/simulation/guides/explore-scenarios' },
+ { title: 'Explore scenario graph', href: '/docs/simulation/guides/explore-scenarios/scenario-graph' },
+ { title: 'Add rows', href: '/docs/simulation/guides/explore-scenarios/add-rows' },
+ { title: 'Add columns', href: '/docs/simulation/guides/explore-scenarios/add-columns' },
]
},
- { title: 'Prompt Simulation', href: '/docs/simulation/features/prompt-simulation' },
- { title: 'Evaluate Tool Calling', href: '/docs/simulation/features/evaluate-tool-calling' },
- { title: 'View Results', href: '/docs/simulation/features/view-results' },
- { title: 'Fix My Agent', href: '/docs/simulation/features/fix-my-agent' },
+ {
+ title: 'Running simulations',
+ items: [
+ { title: 'Create a simulation', href: '/docs/simulation/guides/create-simulation' },
+ { title: 'Run a voice simulation', href: '/docs/simulation/guides/run-voice-simulation' },
+ { title: 'Run a chat simulation', href: '/docs/simulation/guides/run-chat-simulation' },
+ { title: 'Simulate a prompt', href: '/docs/simulation/guides/prompt-simulation' },
+ ]
+ },
+ {
+ title: 'Evaluations in simulation',
+ items: [
+ { title: 'Edit evals in a simulation', href: '/docs/simulation/guides/edit-evals' },
+ { title: 'Evaluate tool calls', href: '/docs/simulation/guides/evaluate-tool-calls' },
+ ]
+ },
+ {
+ title: 'Replay simulations',
+ items: [
+ { title: 'Replay chat sessions', href: '/docs/simulation/guides/replay-chat' },
+ { title: 'Replay voice calls', href: '/docs/simulation/guides/replay-voice' },
+ ]
+ },
+ {
+ title: 'Explore results',
+ items: [
+ { title: 'Overview', href: '/docs/simulation/guides/explore-results' },
+ { title: 'Calls & transcripts', href: '/docs/simulation/guides/explore-results/calls-and-transcripts' },
+ { title: 'Analytics & metrics', href: '/docs/simulation/guides/explore-results/analytics' },
+ ]
+ },
+ { title: 'Fix My Agent', href: '/docs/simulation/guides/fix-my-agent' },
+ {
+ title: 'Optimize using simulate',
+ items: [
+ { title: 'Running optimizations', href: '/docs/simulation/guides/running-optimizations' },
+ { title: 'Optimization runs', href: '/docs/simulation/guides/optimization-runs' },
+ ]
+ },
+ ]
+ },
+ {
+ title: 'References',
+ items: [
+ { title: 'Built-in personas', href: '/docs/simulation/reference/built-in-personas' },
+ { title: 'Voice providers', href: '/docs/simulation/reference/voice-providers' },
+ { title: 'Call metrics', href: '/docs/simulation/reference/call-metrics' },
+ { title: 'SDK & API', href: '/docs/simulation/reference/sdk-api' },
+ ]
+ },
+ {
+ title: 'Troubleshooting',
+ items: [
+ { title: 'Simulation FAQ & fixes', href: '/docs/simulation/troubleshooting' },
]
},
]
@@ -881,7 +1035,6 @@ export const tabNavigation: NavTab[] = [
title: 'Prompt',
items: [
{ title: 'Prompt Versioning: Create, Label, and Serve Prompt Versions', href: '/docs/cookbook/quickstart/prompt-versioning' },
- { title: 'Prototype and Iterate on LLM Applications', href: '/docs/cookbook/quickstart/prototype-llm-app' },
]
},
{
diff --git a/src/lib/redirects.ts b/src/lib/redirects.ts
index 8b6af070..006594e8 100644
--- a/src/lib/redirects.ts
+++ b/src/lib/redirects.ts
@@ -1,6 +1,32 @@
// Auto-generated redirect map: old Mintlify URLs → new docs URLs
// 275 redirects from futureagi.mintlify.app
export const redirectMap: Record = {
+ // Prototype was deprecated and removed from the dashboard; its pages now point at Evaluation
+ '/docs/prototype': '/docs/evaluation',
+ '/docs/prototype/concepts/understanding-prototype': '/docs/evaluation',
+ '/docs/prototype/concepts/versions-and-runs': '/docs/evaluation',
+ '/docs/prototype/features/set-up-prototype': '/docs/evaluation/guides/running-evaluations',
+ '/docs/prototype/features/evals': '/docs/evaluation/builtin',
+ '/docs/prototype/features/choose-winner': '/docs/evaluation',
+ '/docs/cookbook/quickstart/prototype-llm-app': '/docs/cookbook/quickstart/experimentation-compare-prompts',
+ // Error Feed revamp: Concepts/Features replaced by Concepts/Guides/Reference/Troubleshooting.
+ // NOTE /docs/error-feed/features/sampling is linked from inside the product
+ // (frontend ConfigureProject.jsx), so that redirect must not be removed.
+ '/docs/error-feed/concepts/how-it-works': '/docs/error-feed/concepts/understanding-error-feed',
+ '/docs/error-feed/concepts/taxonomy': '/docs/error-feed/reference/error-taxonomy',
+ '/docs/error-feed/concepts/scoring': '/docs/error-feed/concepts/trace-error-analysis',
+ '/docs/error-feed/features/sampling': '/docs/error-feed/guides/turn-on-error-feed',
+ '/docs/error-feed/features/the-feed': '/docs/error-feed/guides/triage-issues',
+ '/docs/error-feed/features/triage-workflow': '/docs/error-feed/guides/triage-issues',
+ '/docs/error-feed/features/issue-overview': '/docs/error-feed/guides/investigate-an-issue',
+ '/docs/error-feed/features/traces': '/docs/error-feed/guides/investigate-an-issue',
+ '/docs/error-feed/features/state-graph': '/docs/error-feed/guides/investigate-an-issue',
+ '/docs/error-feed/features/trends': '/docs/error-feed/guides/investigate-an-issue',
+ '/docs/error-feed/features/metadata-panel': '/docs/error-feed/guides/investigate-an-issue',
+ '/docs/error-feed/features/deep-analysis': '/docs/error-feed/guides/run-root-cause-analysis',
+ '/docs/error-feed/features/linear-integration': '/docs/error-feed/guides/create-linear-issue',
+ '/docs/error-feed/taxonomy': '/docs/error-feed/reference/error-taxonomy',
+
// Manual-instrumentation pages moved from Observe features into the traceAI SDK section
'/docs/observe/features/manual-tracing/set-up-tracing': '/docs/sdk/tracing/set-up-tracing',
'/docs/observe/features/manual-tracing/instrument-with-traceai-helpers': '/docs/sdk/tracing/instrument-with-traceai-helpers',
@@ -18,10 +44,10 @@ export const redirectMap: Record = {
'/docs/observe/features/manual-tracing/langfuse-integration': '/docs/sdk/tracing/langfuse-integration',
'/docs/cookbook/observability': '/docs/cookbook/observe-langgraph-agent-and-obtain-insights',
'/docs/cookbook/improve-langgraph-agent-with-observability': '/docs/cookbook/observe-langgraph-agent-and-obtain-insights',
- '/docs/observe/features/annotation-queue-using-sdk': '/docs/annotations/sdk/annotation-queue-using-sdk',
+ '/docs/observe/features/annotation-queue-using-sdk': '/docs/annotations/reference/sdk-api',
// SDK pages restructured: sdk/tracing.mdx flat page → sdk/tracing/ folder (index still serves /docs/sdk/tracing); annotation-queues moved under Annotations
'/docs/sdk/tracing': '/docs/sdk/tracing/set-up-tracing',
- '/docs/sdk/annotation-queues': '/docs/annotations/sdk/annotation-queue-using-sdk',
+ '/docs/sdk/annotation-queues': '/docs/annotations/reference/sdk-api',
'/docs/observe/voice/set-up': '/docs/observe/features/voice',
'/docs/quickstart/installation': '/docs/installation',
'/docs/observability': '/docs/tracing/auto',
@@ -37,14 +63,16 @@ export const redirectMap: Record = {
'/docs/evaluation/features/futureagi-models': '/docs/evaluation/concepts/evaluator-models',
'/docs/evaluation/concepts/eval-results': '/docs/evaluation/reference/output-types',
'/docs/optimization/optimizers/overview': '/docs/optimization',
- '/docs/dataset/add-annotations': '/docs/dataset/features/annotate',
- '/docs/knowledge-base/concept': '/docs/knowledge-base/concepts/concept',
+ '/docs/dataset/add-annotations': '/docs/annotations/guides/explore-queue/add-items',
+ '/docs/knowledge-base/concept': '/docs/knowledge-base/concepts/understanding-knowledge-base',
'/docs/prompt-workbench': '/docs/prompt',
- '/docs/prompt-workbench/sdk': '/docs/prompt/features/sdk',
+ '/docs/prompt-workbench/sdk': '/docs/prompt/reference/sdk-api',
'/docs/tracing/manual/log-prompt-templates': '/docs/sdk/tracing/log-prompt-templates',
'/docs/tracing/manual/in-line-evals': '/docs/sdk/tracing/in-line-evals',
'/docs/simulation/set-up/scenarios': '/docs/simulation/concepts/scenarios',
- '/docs/simulation/set-up/agent-definition': '/docs/simulation/concepts/agent-definition',
+ '/docs/simulation/set-up/agent-definition': '/docs/simulation/concepts/agent-definitions',
+ '/docs/simulation/concepts/agent-definition': '/docs/simulation/concepts/agent-definitions',
+ '/docs/simulation/concepts/global-nodes': '/docs/simulation/concepts/scenarios',
'/docs/tracing/concepts/components': '/docs/tracing/concepts',
'/docs/tracing/manual/add-attributes-metadata-tags': '/docs/sdk/tracing/add-attributes-metadata-tags',
'/docs/tracing/manual/add-events-exceptions-status': '/docs/sdk/tracing/add-events-exceptions-status',
@@ -186,9 +214,9 @@ export const redirectMap: Record = {
'/future-agi/get-started/evaluation/future-agi-models': '/docs/evaluation/concepts/evaluator-models',
'/future-agi/get-started/evaluation/running-your-first-eval': '/docs/evaluation/guides/running-evaluations',
'/future-agi/get-started/evaluation/use-custom-models': '/docs/evaluation/guides/custom-models',
- '/future-agi/get-started/knowledge-base/concept': '/docs/knowledge-base/concepts/concept',
- '/future-agi/get-started/knowledge-base/how-to/create-kb-using-sdk': '/docs/knowledge-base/features/sdk',
- '/future-agi/get-started/knowledge-base/how-to/create-kb-using-ui': '/docs/knowledge-base/features/ui',
+ '/future-agi/get-started/knowledge-base/concept': '/docs/knowledge-base/concepts/understanding-knowledge-base',
+ '/future-agi/get-started/knowledge-base/how-to/create-kb-using-sdk': '/docs/knowledge-base/guides/manage-with-the-sdk',
+ '/future-agi/get-started/knowledge-base/how-to/create-kb-using-ui': '/docs/knowledge-base/guides/create-knowledge-base',
'/future-agi/get-started/knowledge-base/overview': '/docs/knowledge-base',
'/future-agi/get-started/observability/manual-tracing/add-attributes-metadata-tags': '/docs/sdk/tracing/add-attributes-metadata-tags',
'/future-agi/get-started/observability/manual-tracing/add-events-exceptions-status': '/docs/sdk/tracing/add-events-exceptions-status',
@@ -205,23 +233,23 @@ export const redirectMap: Record = {
'/future-agi/get-started/observability/manual-tracing/set-session-user-id': '/docs/sdk/tracing/set-session-user-id',
'/future-agi/get-started/observability/manual-tracing/set-up-tracing': '/docs/sdk/tracing/set-up-tracing',
'/future-agi/get-started/optimization/dataset-optimization': '/docs/cookbook/quickstart/dataset-optimization',
- '/future-agi/get-started/optimization/how-to/using-python-sdk': '/docs/optimization/features/using-python-sdk',
- '/future-agi/get-started/optimization/optimizers/bayesian-search': '/docs/optimization/optimizers/bayesian-search',
- '/future-agi/get-started/optimization/optimizers/gepa': '/docs/optimization/optimizers/gepa',
- '/future-agi/get-started/optimization/optimizers/meta-prompt': '/docs/optimization/optimizers/meta-prompt',
+ '/future-agi/get-started/optimization/how-to/using-python-sdk': '/docs/optimization/guides/optimize-from-the-sdk',
+ '/future-agi/get-started/optimization/optimizers/bayesian-search': '/docs/optimization/reference/optimizers/bayesian-search',
+ '/future-agi/get-started/optimization/optimizers/gepa': '/docs/optimization/reference/optimizers/gepa',
+ '/future-agi/get-started/optimization/optimizers/meta-prompt': '/docs/optimization/reference/optimizers/meta-prompt',
'/future-agi/get-started/optimization/optimizers/overview': '/docs/optimization',
- '/future-agi/get-started/optimization/optimizers/promptwizard': '/docs/optimization/optimizers/promptwizard',
- '/future-agi/get-started/optimization/optimizers/protegi': '/docs/optimization/optimizers/protegi',
- '/future-agi/get-started/optimization/optimizers/random-search': '/docs/optimization/optimizers/random-search',
+ '/future-agi/get-started/optimization/optimizers/promptwizard': '/docs/optimization/reference/optimizers/promptwizard',
+ '/future-agi/get-started/optimization/optimizers/protegi': '/docs/optimization/reference/optimizers/protegi',
+ '/future-agi/get-started/optimization/optimizers/random-search': '/docs/optimization/reference/optimizers/random-search',
'/future-agi/get-started/optimization/overview': '/docs/optimization',
'/future-agi/get-started/optimization/quickstart': '/docs/optimization',
- '/future-agi/get-started/protect/concept': '/docs/protect/concepts/concept',
- '/future-agi/get-started/protect/how-to': '/docs/protect/features/run-protect',
+ '/future-agi/get-started/protect/concept': '/docs/protect/concepts/understanding-protect',
+ '/future-agi/get-started/protect/how-to': '/docs/protect/guides/run-protect-from-the-sdk',
'/future-agi/get-started/protect/overview': '/docs/protect',
- '/future-agi/get-started/prototype/evals': '/docs/prototype/features/evals',
- '/future-agi/get-started/prototype/overview': '/docs/prototype',
+ '/future-agi/get-started/prototype/evals': '/docs/evaluation/builtin',
+ '/future-agi/get-started/prototype/overview': '/docs/evaluation',
'/future-agi/get-started/prototype/quickstart': '/docs/observe/features/quickstart',
- '/future-agi/get-started/prototype/winner': '/docs/prototype/features/choose-winner',
+ '/future-agi/get-started/prototype/winner': '/docs/evaluation',
'/future-agi/products/observability/auto-instrumentation/overview': '/docs/tracing/auto',
'/future-agi/products/observability/concept/core-components': '/docs/tracing/concepts',
'/future-agi/products/observability/concept/otel': '/docs/tracing/concepts/otel',
@@ -271,58 +299,58 @@ export const redirectMap: Record = {
'/integrations/vertexai': '/docs/integrations/traceai/vertexai',
'/product/agent-compass/overview': '/docs/error-feed',
'/product/agent-compass/quickstart': '/docs/error-feed',
- '/product/agent-compass/taxonomy': '/docs/error-feed/concepts/taxonomy',
+ '/product/agent-compass/taxonomy': '/docs/error-feed/reference/error-taxonomy',
'/docs/cookbook/quickstart/agent-compass-debug': '/docs/error-feed',
- '/product/annotations/concepts/labels': '/docs/annotations/features/labels',
- '/product/annotations/concepts/queues': '/docs/annotations/features/queues',
+ '/product/annotations/concepts/labels': '/docs/annotations/reference/label-types-and-values',
+ '/product/annotations/concepts/queues': '/docs/annotations/reference/queue-settings-and-limits',
'/product/annotations/concepts/scores': '/docs/annotations/concepts/scores',
- '/product/annotations/features/add-items': '/docs/annotations/features/add-items',
- '/product/annotations/features/analytics': '/docs/annotations/features/analytics',
- '/product/annotations/features/annotate': '/docs/annotations/features/annotate',
- '/product/annotations/features/automation': '/docs/annotations/features/automation',
- '/product/annotations/features/export': '/docs/annotations/features/export',
- '/product/annotations/features/inline': '/docs/annotations/features/inline',
- '/product/annotations/features/labels': '/docs/annotations/features/labels',
- '/product/annotations/features/queues': '/docs/annotations/features/queues',
+ '/product/annotations/features/add-items': '/docs/annotations/guides/explore-queue/add-items',
+ '/product/annotations/features/analytics': '/docs/annotations/guides/explore-queue/progress-and-agreement',
+ '/product/annotations/features/annotate': '/docs/annotations/guides/annotate-items',
+ '/product/annotations/features/automation': '/docs/annotations/guides/explore-queue/automate-item-intake',
+ '/product/annotations/features/export': '/docs/annotations/guides/export-annotations',
+ '/product/annotations/features/inline': '/docs/annotations/guides/annotate-without-a-queue',
+ '/product/annotations/features/labels': '/docs/annotations/reference/label-types-and-values',
+ '/product/annotations/features/queues': '/docs/annotations/reference/queue-settings-and-limits',
'/product/annotations/overview': '/docs/annotations',
- '/product/annotations/quickstart': '/docs/annotations/quickstart',
- '/product/annotations/sdk/javascript': '/docs/annotations/sdk/javascript',
- '/product/annotations/sdk/python': '/docs/annotations/sdk/python',
- '/product/dataset/how-to/add-rows-to-dataset': '/docs/dataset/features/add-rows',
- '/product/dataset/how-to/annotate-dataset': '/docs/dataset/features/annotate',
- '/product/dataset/how-to/create-dynamic-column/by-executing-code': '/docs/dataset/features/add-columns',
- '/product/dataset/how-to/create-dynamic-column/by-extracting-entities': '/docs/dataset/features/add-columns',
- '/product/dataset/how-to/create-dynamic-column/by-extracting-json': '/docs/dataset/features/add-columns',
- '/product/dataset/how-to/create-dynamic-column/using-api-calls': '/docs/dataset/features/add-columns',
- '/product/dataset/how-to/create-dynamic-column/using-classification': '/docs/dataset/features/add-columns',
- '/product/dataset/how-to/create-dynamic-column/using-conditional-node': '/docs/dataset/features/add-columns',
- '/product/dataset/how-to/create-dynamic-column/using-run-prompt': '/docs/dataset/features/add-columns',
- '/product/dataset/how-to/create-dynamic-column/using-vector-db': '/docs/dataset/features/add-columns',
- '/product/dataset/how-to/create-new-dataset': '/docs/dataset/features/create',
- '/product/dataset/how-to/create-static-column': '/docs/dataset/features/add-columns',
- '/product/dataset/how-to/experiments-in-dataset': '/docs/dataset/features/experiments',
- '/product/dataset/how-to/run-prompt-in-dataset': '/docs/dataset/features/run-prompt',
+ '/product/annotations/quickstart': '/docs/annotations/guides/create-queue',
+ '/product/annotations/sdk/javascript': '/docs/annotations/reference/sdk-api',
+ '/product/annotations/sdk/python': '/docs/annotations/reference/sdk-api',
+ '/product/dataset/how-to/add-rows-to-dataset': '/docs/dataset/guides/add-rows',
+ '/product/dataset/how-to/annotate-dataset': '/docs/annotations/guides/explore-queue/add-items',
+ '/product/dataset/how-to/create-dynamic-column/by-executing-code': '/docs/dataset/reference/dynamic-column-methods',
+ '/product/dataset/how-to/create-dynamic-column/by-extracting-entities': '/docs/dataset/reference/dynamic-column-methods',
+ '/product/dataset/how-to/create-dynamic-column/by-extracting-json': '/docs/dataset/reference/dynamic-column-methods',
+ '/product/dataset/how-to/create-dynamic-column/using-api-calls': '/docs/dataset/reference/dynamic-column-methods',
+ '/product/dataset/how-to/create-dynamic-column/using-classification': '/docs/dataset/reference/dynamic-column-methods',
+ '/product/dataset/how-to/create-dynamic-column/using-conditional-node': '/docs/dataset/reference/dynamic-column-methods',
+ '/product/dataset/how-to/create-dynamic-column/using-run-prompt': '/docs/dataset/reference/dynamic-column-methods',
+ '/product/dataset/how-to/create-dynamic-column/using-vector-db': '/docs/dataset/reference/dynamic-column-methods',
+ '/product/dataset/how-to/create-new-dataset': '/docs/dataset/guides/create-a-dataset',
+ '/product/dataset/how-to/create-static-column': '/docs/dataset/reference/dynamic-column-methods',
+ '/product/dataset/how-to/experiments-in-dataset': '/docs/dataset/guides/run-an-experiment',
+ '/product/dataset/how-to/run-prompt-in-dataset': '/docs/dataset/guides/run-a-prompt-on-every-row',
'/product/dataset/overview': '/docs/dataset',
- '/product/prompt/how-to/create-prompt-from-existing-template': '/docs/prompt/features/create-from-template',
- '/product/prompt/how-to/create-prompt-from-scratch': '/docs/prompt/features/create-from-scratch',
- '/product/prompt/how-to/linked-traces': '/docs/prompt/features/linked-traces',
- '/product/prompt/how-to/manage-folders': '/docs/prompt/features/folders',
- '/product/prompt/how-to/prompt-workbench-using-sdk': '/docs/prompt/features/sdk',
+ '/product/prompt/how-to/create-prompt-from-existing-template': '/docs/prompt/guides/create-a-prompt',
+ '/product/prompt/how-to/create-prompt-from-scratch': '/docs/prompt/guides/create-a-prompt',
+ '/product/prompt/how-to/linked-traces': '/docs/prompt/guides/track-prompt-performance',
+ '/product/prompt/how-to/manage-folders': '/docs/prompt/guides/organize-prompts-in-folders',
+ '/product/prompt/how-to/prompt-workbench-using-sdk': '/docs/prompt/reference/sdk-api',
'/product/prompt/overview': '/docs/prompt',
- '/product/simulation/agent-definition': '/docs/simulation/concepts/agent-definition',
- '/product/simulation/how-to/chat-simulation-using-sdk': '/docs/simulation/features/simulation-using-sdk',
- '/product/simulation/how-to/evaluate-tool-calling': '/docs/simulation/features/evaluate-tool-calling',
- '/product/simulation/how-to/fix-my-agent': '/docs/simulation/features/fix-my-agent',
- '/product/simulation/how-to/observe-to-simulate': '/docs/simulation/features/observe-to-simulate',
- '/product/simulation/how-to/prompt-simulation': '/docs/simulation/features/prompt-simulation',
- '/product/simulation/how-to/voice-observability': '/docs/simulation/features/voice-replay',
+ '/product/simulation/agent-definition': '/docs/simulation/concepts/agent-definitions',
+ '/product/simulation/how-to/chat-simulation-using-sdk': '/docs/simulation/guides/run-chat-simulation',
+ '/product/simulation/how-to/evaluate-tool-calling': '/docs/simulation/guides/evaluate-tool-calls',
+ '/product/simulation/how-to/fix-my-agent': '/docs/simulation/guides/fix-my-agent',
+ '/product/simulation/how-to/observe-to-simulate': '/docs/simulation/guides/replay-chat',
+ '/product/simulation/how-to/prompt-simulation': '/docs/simulation/guides/prompt-simulation',
+ '/product/simulation/how-to/voice-observability': '/docs/simulation/guides/replay-voice',
'/product/simulation/overview': '/docs/simulation',
'/product/simulation/personas': '/docs/simulation/concepts/personas',
- '/product/simulation/run-tests': '/docs/simulation/features/run-simulation',
+ '/product/simulation/run-tests': '/docs/simulation/guides/run-voice-simulation',
'/product/simulation/scenarios': '/docs/simulation/concepts/scenarios',
'/quickstart/generate-synthetic-data': '/docs/quickstart/generate-synthetic-data',
'/quickstart/running-evals-in-simulation': '/docs/quickstart/running-evals-in-simulation',
- '/quickstart/setup-mcp-server': '/docs/quickstart/setup-mcp-server',
+ '/quickstart/setup-mcp-server': '/docs/falcon-ai/guides/use-the-mcp-server',
'/quickstart/setup-observability': '/docs/quickstart/setup-observability',
'/release-notes': '/docs/release-notes',
'/sdk-reference/datasets': '/docs/sdk/datasets',
@@ -334,4 +362,65 @@ export const redirectMap: Record = {
'/sdk-reference/testcase': '/docs/sdk/testcase',
'/sdk-reference/tracing': '/docs/sdk/tracing',
'/docs/self-hosting/environment': '/docs/self-hosting/configuration/environment',
+ '/docs/simulation/features/run-simulation': '/docs/simulation/guides/run-voice-simulation',
+ '/docs/simulation/features/simulation-using-sdk': '/docs/simulation/guides/run-chat-simulation',
+ '/docs/simulation/features/observe-to-simulate': '/docs/simulation/guides/replay-chat',
+ '/docs/simulation/features/voice-replay': '/docs/simulation/guides/replay-voice',
+ '/docs/simulation/features/prompt-simulation': '/docs/simulation/guides/prompt-simulation',
+ '/docs/simulation/features/evaluate-tool-calling': '/docs/simulation/guides/evaluate-tool-calls',
+ '/docs/simulation/features/view-results': '/docs/simulation/guides/explore-results',
+ '/docs/simulation/features/fix-my-agent': '/docs/simulation/guides/fix-my-agent',
+
+ // Product docs revamp: Features-shaped sections replaced by Concepts/Guides/Reference/Troubleshooting.
+ '/docs/agent-playground/features/build-workflow': '/docs/agent-playground/guides/build-workflow',
+ '/docs/agent-playground/features/create-graph': '/docs/agent-playground/guides/create-agent',
+ '/docs/agent-playground/features/run-and-monitor': '/docs/agent-playground/guides/run-an-agent',
+ '/docs/annotations/features/add-items': '/docs/annotations/guides/explore-queue/add-items',
+ '/docs/annotations/features/analytics': '/docs/annotations/guides/explore-queue/progress-and-agreement',
+ '/docs/annotations/features/annotate': '/docs/annotations/guides/annotate-items',
+ '/docs/annotations/features/automation': '/docs/annotations/guides/explore-queue/automate-item-intake',
+ '/docs/annotations/features/export': '/docs/annotations/guides/export-annotations',
+ '/docs/annotations/features/inline': '/docs/annotations/guides/annotate-without-a-queue',
+ '/docs/annotations/features/labels': '/docs/annotations/reference/label-types-and-values',
+ '/docs/annotations/features/queues': '/docs/annotations/reference/queue-settings-and-limits',
+ '/docs/annotations/quickstart': '/docs/annotations/guides/create-queue',
+ '/docs/annotations/sdk/annotation-queue-using-sdk': '/docs/annotations/reference/sdk-api',
+ '/docs/annotations/sdk/javascript': '/docs/annotations/reference/sdk-api',
+ '/docs/annotations/sdk/python': '/docs/annotations/reference/sdk-api',
+ '/docs/dataset/concept/dynamic-column': '/docs/dataset/concepts/static-and-dynamic-columns',
+ '/docs/dataset/concept/static-column': '/docs/dataset/concepts/static-and-dynamic-columns',
+ '/docs/dataset/concept/synthetic-data': '/docs/dataset/concepts/synthetic-data',
+ '/docs/dataset/concept/understanding-dataset': '/docs/dataset/concepts/understanding-datasets',
+ '/docs/dataset/features/add-columns': '/docs/dataset/reference/dynamic-column-methods',
+ '/docs/dataset/features/add-rows': '/docs/dataset/guides/add-rows',
+ '/docs/dataset/features/annotate': '/docs/annotations/guides/explore-queue/add-items',
+ '/docs/dataset/guides/annotate-rows': '/docs/annotations/guides/explore-queue/add-items',
+ '/docs/dataset/features/create': '/docs/dataset/guides/create-a-dataset',
+ '/docs/dataset/features/experiments': '/docs/dataset/guides/run-an-experiment',
+ '/docs/dataset/features/run-prompt': '/docs/dataset/guides/run-a-prompt-on-every-row',
+ '/docs/falcon-ai/features/chat': '/docs/falcon-ai/guides/chat-with-falcon-ai',
+ '/docs/falcon-ai/features/mcp-connectors': '/docs/falcon-ai/concepts/mcp-connectors',
+ '/docs/falcon-ai/features/skills': '/docs/falcon-ai/concepts/skills',
+ '/docs/knowledge-base/concepts/concept': '/docs/knowledge-base/concepts/understanding-knowledge-base',
+ '/docs/knowledge-base/features/sdk': '/docs/knowledge-base/guides/manage-with-the-sdk',
+ '/docs/knowledge-base/features/ui': '/docs/knowledge-base/guides/create-knowledge-base',
+ '/docs/optimization/concepts/concept': '/docs/optimization/concepts/understanding-optimization',
+ '/docs/optimization/features/using-platform': '/docs/optimization/guides/run-an-optimization',
+ '/docs/optimization/features/using-python-sdk': '/docs/optimization/guides/optimize-from-the-sdk',
+ '/docs/optimization/optimizers/bayesian-search': '/docs/optimization/reference/optimizers/bayesian-search',
+ '/docs/optimization/optimizers/gepa': '/docs/optimization/reference/optimizers/gepa',
+ '/docs/optimization/optimizers/meta-prompt': '/docs/optimization/reference/optimizers/meta-prompt',
+ '/docs/optimization/optimizers/promptwizard': '/docs/optimization/reference/optimizers/promptwizard',
+ '/docs/optimization/optimizers/protegi': '/docs/optimization/reference/optimizers/protegi',
+ '/docs/optimization/optimizers/random-search': '/docs/optimization/reference/optimizers/random-search',
+ '/docs/prompt/features/create-from-scratch': '/docs/prompt/guides/create-a-prompt',
+ '/docs/prompt/features/create-from-template': '/docs/prompt/guides/create-a-prompt',
+ '/docs/prompt/features/create-with-ai': '/docs/prompt/guides/create-a-prompt',
+ '/docs/prompt/features/folders': '/docs/prompt/guides/organize-prompts-in-folders',
+ '/docs/prompt/features/linked-traces': '/docs/prompt/guides/track-prompt-performance',
+ '/docs/prompt/features/sdk': '/docs/prompt/reference/sdk-api',
+ '/docs/protect/concepts/concept': '/docs/protect/concepts/understanding-protect',
+ '/docs/protect/concepts/guardrail-pipeline': '/docs/protect/concepts/understanding-protect',
+ '/docs/protect/features/run-protect': '/docs/protect/guides/run-protect-from-the-sdk',
+ '/docs/quickstart/setup-mcp-server': '/docs/falcon-ai/guides/use-the-mcp-server',
};
diff --git a/src/pages/docs/agent-playground/concepts/understanding-agent-playground.mdx b/src/pages/docs/agent-playground/concepts/understanding-agent-playground.mdx
index 5bf045fd..4834c5bc 100644
--- a/src/pages/docs/agent-playground/concepts/understanding-agent-playground.mdx
+++ b/src/pages/docs/agent-playground/concepts/understanding-agent-playground.mdx
@@ -1,100 +1,69 @@
---
-title: "Agent Playground: Core Concepts"
-description: "Learn the core building blocks of Agent Playground: graphs, LLM nodes, subgraph nodes, ports, edges, and node templates."
+title: "Understanding Agent Playground"
+description: "How nodes, connections, and nested agents combine into what an agent actually runs"
---
-## About
+## An agent is a set of connected nodes
-Agent Playground is built around a small set of core building blocks. Understanding these helps you design and debug workflows effectively. This page explains how graphs, nodes, ports, edges, and templates fit together.
+An **agent** is a set of nodes wired together. Each node takes named inputs and produces named outputs, and every input and output is typed: it carries a display name for the canvas and a JSON Schema that defines the shape of the data passing through it. A **connection** joins one node's output to another node's input, and that's how a value produced by one step reaches the next.
----
-
-## Graphs
-
-A graph is the top-level container for your AI workflow. It is a series of connected steps where data flows from inputs through each node to outputs.
+Take an agent called `support-triage`. It starts with two nodes: `classify` and `draft-reply`. `classify` takes a `message` input and produces a `response` output: it runs a linked [prompt](/docs/prompt/concepts/understanding-prompts), and that's what turns the input into the output. `classify`'s `response` output schema tracks whatever prompt is linked to it: change the linked prompt's response format, and `response`'s shape updates to match. A connection carries that `classify.response` value straight into `draft-reply`'s own `category` input.
-Each graph has:
-- **Name and description** for identification
-- **Collaborators** who can view and edit the graph
-- **One or more versions** (snapshots of the workflow at different points in time)
+Both `classify` and `draft-reply` are atomic nodes, meaning each does the work itself rather than delegating to another agent; right now that means both are LLM Prompt nodes, since that's the only kind of atomic node the platform ships with.
----
-
-## Nodes
+A node doesn't have to do the work itself, either. It can instead be a reference to another agent's [saved version](/docs/agent-playground/concepts/versions-and-execution), a version you've saved rather than a draft still being edited, letting you reuse a whole agent as a single step, as covered in Composing agents with an Agent Node below.
-Nodes are the building blocks of your workflow. Each node represents a single step that takes inputs, performs an operation, and produces outputs.
+## How connections wire together
-### LLM Prompt Nodes
+Three properties hold for every connection in an agent, and they explain most of what you'll run into.
-LLM Prompt nodes execute a prompt against a language model. They connect directly to the **Prompt Management** system:
+- **One output can feed several inputs at once.** If `classify`'s `response` output is useful to more than one downstream node, connect it to as many inputs as you need; each one gets the same value
+- **Every input accepts exactly one connection.** `draft-reply`'s `category` input can be fed by `classify` or by some other node, but never both at the same time. If two outputs could plausibly feed the same input, you pick one
+- **The wiring can never loop back on itself.** Data only flows forward, from a node to the nodes downstream of it, never back to a node it already came from. An agent that tried to connect `draft-reply`'s output back into `classify`'s input would be forming a loop, and the platform rejects that connection
-- **Prompt template** defines the prompt text with `{{variable}}` placeholders
-- **Model** specifies which LLM to call (GPT-4, Claude, etc.)
-- **Parameters** control generation behavior (temperature, max tokens, top-p)
-- **Response format** determines output structure (plain text or JSON)
+## What isn't connected becomes the agent's own input or output
-When the linked prompt template is updated, the node's input ports automatically sync to match the new variables.
+Not every input ends up fed by a connection, and not every output ends up feeding one, and that's what makes the model click.
-### Agent (Subgraph) Nodes
+Inputs that nothing feeds are the agent's own inputs: the [values you fill in before a run](/docs/agent-playground/guides/build-workflow/set-input-variables). `classify`'s `message` input has no connection into it, so `message` is what you provide when you run `support-triage`.
-Agent nodes embed an entire other graph as a single step in your workflow. This enables:
+Outputs that nothing consumes work the same way in reverse: they're what the run hands back. If `draft-reply`'s `response` output isn't wired into anything, `draft-reply`'s `response` is part of `support-triage`'s result.
-- **Modularity**: break complex workflows into reusable sub-workflows
-- **Composition**: combine multiple agents into a larger pipeline
-- **Encapsulation**: the parent graph only sees the subgraph's exposed input and output ports
+## Composing agents with an Agent Node
-
- Subgraph nodes can only reference **non-draft** versions of other graphs, never drafts. Circular references (Graph A embeds Graph B which embeds Graph A) are detected and blocked.
-
+An **Agent Node** is a node whose job is to run another saved agent as a single step, instead of doing the work itself. You point it at one of that other agent's saved versions, and because the referenced agent has its own inputs, the Agent Node exposes those same inputs as its own. You map each of them in the node's Input Mapping section: it's where you choose which value in the current agent feeds an input that belongs to the nested agent, though you don't have to map every one. Leave a mapping empty and that input becomes one of the agent's own inputs, the same rule covered above. See [Configure an Agent node](/docs/agent-playground/guides/build-workflow/configure-an-agent-node) for the form.
----
+Say `support-triage` needs to compress `draft-reply`'s output before it goes out, using a category-aware summarizer someone already built. Add an Agent Node, `summarize-step`, pointing at a saved version of a separate agent called `summarizer`. `summarizer` takes two inputs, `text` and `category`, and produces one output, `response`. Once `summarize-step` is in place, `text` and `category` become inputs on `summarize-step` itself: `classify`'s `response` output can now fan out to feed both `draft-reply` and `summarize-step`, and `draft-reply`'s `response` output connects into `summarize-step`'s `text` input.
-## Ports
+ CL["classify"]
+ CL -->|"response"| DR["draft-reply"]
+ CL -->|"response"| SS["summarize-step"]
+ DR -->|"response"| SS
+ SS -.->|"points to"| SUM["summarizer (saved version)"]
+ SS -->|"response"| RES(("agent output"))`} />
-Ports are typed connection points on every node. They define the data contract: what a node expects as input and what it produces as output.
-
-Each port has:
-- **Direction**: input or output
-- **Key**: a unique identifier (e.g., `prompt`, `response`, `output`)
-- **Display name**: a human-readable label
-- **Data schema**: a JSON Schema definition that validates data at runtime
-
-### Exposed Ports
-
-When an input port has no incoming edge, it becomes an **exposed port**: an entry point for the graph. Similarly, output ports with no outgoing edges are exposed as graph outputs. Exposed input ports automatically become columns in the graph's dataset for execution.
-
----
+That last connection changes what's exposed. `draft-reply`'s `response` is no longer unconnected, so it drops out of `support-triage`'s result, and `summarize-step`'s own `response` output takes its place as the new exposed output. Nothing else changes: the rest of the wiring, and the rules that govern it, are exactly the ones from the last two sections.
-## Edges
+What an Agent Node can't point at:
-Edges are the connections that carry data between nodes. Each edge links one node's output port to another node's input port.
-
-**Rules:**
-- **Fan-out is allowed**: one output port can connect to multiple input ports (data is broadcast to all targets)
-- **Fan-in is blocked**: each input port accepts only one incoming edge
-- **No cycles**: the graph cannot loop back on itself. The platform detects and prevents cycles at connection time
-- **Type validation**: the platform checks that connected ports have compatible data schemas
-
----
-
-## Node Templates
-
-Node templates are the registry of available node types. They define the default configuration for each type of node, including:
-
-- **Port definitions**: what inputs and outputs the node type has
-- **Port mode**: strict, extensible, or dynamic
-- **Config schema**: JSON Schema for the node's configuration (model parameters, settings, etc.)
-
-The platform ships with built-in templates (LLM Prompt, Agent) and supports custom templates for specialized use cases. Templates are seeded system-wide and available to all users.
-
-
- When you drag a node from the selection panel onto the canvas, the platform creates a new node instance from the matching template and auto-generates its ports based on the template's port definitions.
-
-
----
+- **A draft version.** Only a saved version of the other agent is a valid target
+- **This agent itself.** An agent can't be a step inside itself
+- **An agent that already contains this one as one of its own steps, however indirectly.** That would form a loop between agents instead of within one, and it's rejected the same way a loop inside a single agent is
-## Next Steps
+## Keep exploring
-- [Versions & Execution](/docs/agent-playground/concepts/versions-and-execution): How the version lifecycle and execution model work
-- [Create a Graph](/docs/agent-playground/features/create-graph): Create your first workflow
-- [Build a Workflow](/docs/agent-playground/features/build-workflow): Add nodes, configure them, and connect them
+
+
+ How a draft becomes a saved version, and how a run moves data through an agent
+
+
+ Create the agent itself, before you add any nodes
+
+
+ Add nodes to that agent, configure them, and wire the connections between them
+
+
diff --git a/src/pages/docs/agent-playground/concepts/versions-and-execution.mdx b/src/pages/docs/agent-playground/concepts/versions-and-execution.mdx
index 8cfc810f..5bc94d76 100644
--- a/src/pages/docs/agent-playground/concepts/versions-and-execution.mdx
+++ b/src/pages/docs/agent-playground/concepts/versions-and-execution.mdx
@@ -1,100 +1,65 @@
---
-title: "Agent Workflow Versions & Execution"
-description: "Understand Agent Playground's draft/non-draft version lifecycle, topological execution model, node states, and data routing between nodes."
+title: "Versions & Execution"
+description: "How drafts become saved versions, and what one run records"
---
-## About
+## Versions and executions: two models
-This page covers how Agent Playground handles versioning and execution. Versions let you iterate safely on your workflow. Execution is how the platform runs your graph and tracks results per node.
+A version is a numbered snapshot of an agent's graph. A version starts as a **draft**, the state you can edit; saving fixes it into an immutable snapshot. An **execution** is the record of one run of a version.
----
-
-## Version Lifecycle
-
-Every graph manages its workflow through **versions**: immutable snapshots of the graph's structure (nodes, ports, edges, and configuration) at a point in time.
-
-### Draft vs Non-Draft
-
-There are two states a version can be in:
-
-| State | Editable | Executable |
-|-------|----------|------------|
-| **Draft** | Yes | No |
-| **Non-draft** | No | Yes |
-
-### How Versions Work
-
-1. **Draft** - When you create or modify a graph, you work in a draft. Drafts are auto-saved as you make changes. This is your workspace for experimenting and iterating freely.
-
-2. **Save** - When a draft is ready, you save it. This finalizes the draft into a non-draft version, making it the version that runs when you execute the graph. The previous version is kept in your version history.
+A run always executes one specific saved version, so the three stay linked: what you can currently change, what you saved, and what happened when it ran.
-You can always view older versions in the **Changelog** tab and create a new draft from any of them to pick up where you left off.
+ Versions
+ subgraph Versions["support-triage's versions"]
+ History["Version 1, 2, 3 ..."]
+ Draft["Draft"]
+ Draft -->|"Save Agent"| V4["Version 4 (runs)"]
+ end
+ V4 -->|"Run"| ExecBox
+ subgraph ExecBox["Execution (one run)"]
+ Classify["classify: success"]
+ DraftReply["draft-reply: success"]
+ Summarize["summarize-step: success"]
+ end
+ Summarize --> Nested["Nested execution"]`} />
-
- Saving a version triggers validation: the platform checks that all required connections exist and the graph has no cycles. If validation fails, the version stays as a draft.
-
+## Drafts and versions
-### Version Workflow
+Every change you make, adding a node, editing a connection, rewriting an input, lives in a draft. A draft is marked with the Draft badge, and it is the only kind of version you can edit.
-```
-Create Graph → Draft v1
- ↓ (save)
- v1 → Edit → Draft v2
- ↓ (save)
- v1 v2 → Edit → Draft v3
- ↓ (save)
- v1 v2 v3
-```
+[Save Agent](/docs/agent-playground/guides/build-workflow) turns the draft into a version: you write a commit message for it, and it becomes a numbered snapshot carrying that message. Saving validates the graph before that version can run:
----
-
-## Execution Model
-
-When you run a graph, the platform creates a **graph execution**: a record of that specific run with its own ID, status, timing, and results.
+- [Exposed output](/docs/agent-playground/concepts/understanding-agent-playground#what-isnt-connected-becomes-the-agents-own-input-or-output) names cannot duplicate
+- Every node's required inputs must be present
-### Execution States
+The save is blocked until both checks pass. Once it succeeds, that new version becomes the one the agent runs, replacing whichever version ran before it.
-| State | Meaning |
-|-------|---------|
-| **Pending** | Execution created, waiting to start |
-| **Running** | Nodes are actively executing |
-| **Success** | All nodes completed successfully |
-| **Failed** | One or more nodes encountered an error |
-| **Cancelled** | Execution was stopped by the user |
+Older versions do not disappear. They stay available to read and preview, though only the draft can be edited. An agent always keeps at least one version, so there is never a state with nothing to run.
-### Node Execution
+## Executions and node records
-Within a graph execution, each node gets its own **node execution** record tracking:
-- Start and end timestamps
-- Status (pending, running, success, failed, skipped)
-- Input data received from upstream nodes
-- Output data produced
-- Error details (if failed)
+Running an agent creates an execution: one record of that run as a whole, plus one record per node inside it. Each node record carries its own status, one of pending, running, success, failed, or skipped, along with the inputs it received and the outputs it produced.
-Nodes that cannot execute because an upstream node failed are marked as **skipped**.
-
----
+A node whose upstream step failed is marked skipped rather than being run at all, so a failure does not silently propagate as if the node had executed. In a run of `support-triage`, for example, if `draft-reply` fails, `summarize-step` is skipped rather than run, since it depends on `draft-reply`'s output, while `classify`, which has no dependency on `draft-reply`, is unaffected.
-## Data Routing
+Independent branches do not wait on each other: up to ten nodes run at the same time by default, so parts of the graph with no dependency between them finish in parallel instead of one after another.
-The execution engine processes nodes in **topological order**: it determines which nodes can run first (those with no dependencies) and works forward through the graph.
+Nesting closes the loop between the two models. `summarize-step`, for example, is an [Agent Node](/docs/agent-playground/concepts/understanding-agent-playground#composing-agents-with-an-agent-node): its record holds the nested run of the agent it points at. You can open a run inside a run and keep going as deep as the graph nests.
-### How Data Flows
-
-1. **Graph inputs** are injected into the exposed input ports (ports with no incoming edges)
-2. **Start nodes** (nodes with all inputs satisfied) execute first
-3. When a node completes, its output data is **routed** along edges to downstream nodes' input ports
-4. A downstream node becomes **ready** when all its required input ports have data
-5. Ready nodes execute, and the process repeats until all nodes are done
-6. **Graph outputs** are collected from exposed output ports (ports with no outgoing edges)
-
-Each piece of data flowing through a port is validated against the port's JSON Schema. Validation errors are recorded but do not block execution: you can inspect them after the run to identify data contract issues.
-
----
+## Why the version matters
-## Next Steps
+A run doesn't just execute "the agent", it pins one specific saved version, and each execution stays tied to the version it ran. So two runs of the same agent differ only by what you changed between the versions they pinned.
-- [Create a Graph](/docs/agent-playground/features/create-graph): Create your first workflow and manage versions
-- [Build a Workflow](/docs/agent-playground/features/build-workflow): Add nodes, configure them, and connect them
-- [Run & Monitor](/docs/agent-playground/features/run-and-monitor): Execute workflows and inspect results
+## Keep exploring
+
+
+ Open the Changelog to read and preview older versions
+
+
+ Run a workflow and read node status and output from the Executions tab
+
+
diff --git a/src/pages/docs/agent-playground/features/build-workflow.mdx b/src/pages/docs/agent-playground/features/build-workflow.mdx
deleted file mode 100644
index e2aff337..00000000
--- a/src/pages/docs/agent-playground/features/build-workflow.mdx
+++ /dev/null
@@ -1,126 +0,0 @@
----
-title: "Build an AI Agent Workflow"
-description: "Add LLM Prompt and Agent nodes to the canvas, configure models and parameters, draw edges, and set global variables in Agent Playground."
----
-
-## About
-
-The Agent Builder is the visual graph editor where you assemble your workflow by adding nodes, configuring them, and connecting them with edges. For background on nodes, ports, and edges, see [Understanding Agent Playground](/docs/agent-playground/concepts/understanding-agent-playground).
-
-
-
-The workspace has three main areas:
-- **Node Selection Panel** (left): available node types to add
-- **Canvas** (center): the graph editor where you arrange and connect nodes
-- **Node Drawer** (right): configuration form for the selected node
-
----
-
-## Add Nodes
-
-The left panel shows the available node types. You can add nodes in two ways:
-
-- **Click** a node type to add it to the center of the canvas
-- **Drag** a node type onto the canvas and drop it at the desired position
-
-
-
-### Available Node Types
-
-| Node Type | Purpose |
-|-----------|---------|
-| **LLM Prompt** | Execute a prompt against a language model. Configured via Prompt Templates. |
-| **Agent** | Embed another graph as a sub-workflow for modular composition. |
-
-When you add a node, the platform automatically creates its ports based on the node template's definitions.
-
----
-
-## Configure Nodes
-
-Click any node on the canvas to open the **Node Drawer** on the right side. The drawer shows a configuration form specific to the node type.
-
-
-
-### LLM Prompt Node Configuration
-
-| Field | Description |
-|-------|-------------|
-| **Prompt Template** | Select a prompt template from Prompt Management. The node's input ports automatically sync to the template's `{{variables}}`. |
-| **Model** | Choose the LLM to call (e.g., GPT-4, Claude, Gemini). |
-| **Temperature** | Controls randomness (0 = deterministic, 1 = creative). |
-| **Max Tokens** | Maximum length of the generated response. |
-| **Top-p** | Nucleus sampling threshold. |
-| **Response Format** | Output as plain text or structured JSON. |
-
-
- When you change the prompt template, the node's input ports update automatically to match the new template variables. Existing connections to removed variables are disconnected.
-
-
-### Agent Node Configuration
-
-For Agent (subgraph) nodes, configure:
-- **Agent** - choose which graph to embed as a sub-agent
-- **Version** - select which version of that agent to use
-- **Input mapping** - map variables from the parent graph to the sub-agent's exposed input ports
-
-
-
----
-
-## Connect Nodes
-
-Create data flow connections by drawing edges between nodes.
-
-
-
- Hover over a node's **output handle** (the circle on the right side of the node). Your cursor changes to a crosshair.
-
-
- Click and drag from the output handle toward the target node's **input handle** (the circle on the left side).
-
-
-
- Release the mouse over the target node's input handle. The platform creates the edge and validates that the port types are compatible.
-
-
-
-### Connection Rules
-
-- **One output to many inputs**: an output port can connect to multiple input ports (data is broadcast)
-- **One input, one source**: each input port accepts only one incoming edge
-- **No cycles**: the platform prevents connections that would create loops in the graph
-- **Type checking**: connected ports must have compatible data schemas
-
-To **delete an edge**, select it and press Delete.
-
----
-
-## Global Variables
-
-Use the **Global Variables** panel (accessible from the right side of the builder) to define values for your prompt variables. These are the inputs your workflow needs to run , for example the customer message that gets passed into your first node.
-
-
-
----
-
-## Tips
-
-
- Use the **+** button that appears below a node (when it has no outgoing edge) to quickly add and connect a new node in one step.
-
-
-
- Changes to a draft version are auto-saved as you work. You do not need to manually save after every edit. The platform saves node positions, configurations, and connections automatically.
-
-
-
- If you are viewing a non-draft version, the canvas is read-only. You must create a draft to make changes. The platform will prompt you to create a draft when you try to edit.
-
-
----
-
-## Next Steps
-
-- [Run & Monitor](/docs/agent-playground/features/run-and-monitor): Execute your workflow and watch results in real time
-- [Versions & Execution](/docs/agent-playground/concepts/versions-and-execution): Understand version lifecycle and execution model
diff --git a/src/pages/docs/agent-playground/features/create-graph.mdx b/src/pages/docs/agent-playground/features/create-graph.mdx
deleted file mode 100644
index 0f839633..00000000
--- a/src/pages/docs/agent-playground/features/create-graph.mdx
+++ /dev/null
@@ -1,68 +0,0 @@
----
-title: "Create an Agent Graph & Manage Versions"
-description: "Create a new agent graph in Agent Playground, set metadata, manage draft and saved versions, and roll back to previous workflow snapshots."
----
-
-## About
-
-Create a new graph to start building your AI workflow. A graph is the container for your entire pipeline. For background on what graphs are, see [Understanding Agent Playground](/docs/agent-playground/concepts/understanding-agent-playground).
-
----
-
-## Create a New Graph
-
-
-
- Go to **Agent Playground** from the main navigation. You will see the agent list view showing all your existing graphs.
-
- 
-
-
- Click **Create Agent** in the top-right corner. The platform creates a new graph with a blank draft version and takes you directly to the builder canvas.
-
-
-
- Give your graph a meaningful name and description.
-
-
-
----
-
-## Manage Versions
-
-Agent Playground uses a version system to track changes and let you roll back safely. Every graph starts with a draft version.
-
-### View Versions
-
-Switch to the **Changelog** tab to see all versions of your graph. The left sidebar lists every version with its status and creation date. Click a version to preview its workflow structure in read-only mode on the right panel.
-
-
-
-### Activate a Draft
-
-When your draft is ready for use:
-
-1. Click **Save**
-
-The platform validates the graph (checking for cycles, missing connections, and incomplete configurations). If validation passes, the draft is saved and becomes the version that runs when you execute the graph. The previous version is kept in your version history.
-
-### Create a Draft from a Previous Version
-
-To iterate on an older version:
-
-1. Open the **Changelog** tab
-2. Select any previous version
-3. Click **Create Draft**
-
-This creates a new draft that is a copy of that version's workflow structure. You can then modify it freely without affecting the saved version.
-
-### Edit a Non-Draft Version
-
-If you try to edit a node while viewing a non-draft version, the platform prompts you to create a draft first. Your edits go into the new draft, leaving the saved version unchanged until you explicitly save the draft.
-
----
-
-## Next Steps
-
-- [Build a Workflow](/docs/agent-playground/features/build-workflow): Add nodes, configure them, and connect them into a pipeline
-- [Run & Monitor](/docs/agent-playground/features/run-and-monitor): Execute your workflow and inspect results
diff --git a/src/pages/docs/agent-playground/features/run-and-monitor.mdx b/src/pages/docs/agent-playground/features/run-and-monitor.mdx
deleted file mode 100644
index 97e0f06f..00000000
--- a/src/pages/docs/agent-playground/features/run-and-monitor.mdx
+++ /dev/null
@@ -1,105 +0,0 @@
----
-title: "Run & Monitor Agent Workflows"
-description: "Execute AI agent workflows, watch per-node status in real time, inspect input/output data for each step, and browse full execution history."
----
-
-## About
-
-Run your workflow and monitor each step as it executes. The platform shows real-time status per node, records full input/output data, and keeps a history of all past runs.
-
----
-
-## Run a Workflow
-
-
-
- Navigate to your graph and open the **Build** tab. Make sure all nodes are configured. The platform highlights unconfigured nodes with a red border.
-
-
- Click the **Run** button (play icon) in the builder actions on the right side of the canvas.
-
- - **If you are on a draft version**: the platform validates the graph, prompts you to save, and then activates and executes the workflow.
- - **If you are on a non-draft version**: the workflow executes immediately.
-
- Validation checks for:
- - All nodes are fully configured
- - No cycles in the graph
- - All required ports are connected
-
-
- The **Run Agent Panel** opens at the bottom of the builder. Nodes update in real time as they execute:
-
- - **Green animated border**: node is currently running
- - **Green solid border**: node completed successfully
- - **Red border**: node failed
- - **Gray**: node is pending or was skipped
-
- Edges animate to show data flowing between nodes.
-
- 
-
-
-
----
-
-## View Execution Results
-
-The **Run Agent Panel** at the bottom of the builder shows detailed results after (and during) execution.
-
-
-
-The panel is split into two halves:
-
-### Left: Graph Visualization
-A miniature view of your graph with nodes colored by execution status:
-- Green = success
-- Red = failed
-- Gray = pending or skipped
-
-Click any node in this view to inspect its details on the right.
-
-### Right: Node Output Details
-Shows the selected node's execution data:
-
-| Field | Description |
-|-------|-------------|
-| **Execution ID** | Unique identifier for this node's execution |
-| **Status** | Success, failed, skipped, running, or pending |
-| **Duration** | How long the node took to execute |
-| **Input Data** | The data received from upstream nodes |
-| **Output Data** | The data produced by this node (JSON or text) |
-| **Error** | Error message and details (if the node failed) |
-
-The panel auto-selects the last executed node when the workflow completes.
-
----
-
-## Execution History
-
-The **Executions** tab shows a complete history of all runs for this graph.
-
-
-
-### Browse Executions
-
-The left sidebar lists all executions, most recent first. Each entry shows:
-- Timestamp
-- Status badge (success, failed, running, pending)
-- Version used
-
-Click an execution to load its details on the right. The same graph visualization and node output panel from the builder.
-
-### Inspect a Past Execution
-
-Select any execution to see:
-1. The full graph with per-node status colors
-2. Click individual nodes to see their input data, output data, timing, and errors
-3. Compare different executions to understand how changes affected results
-
----
-
-## Next Steps
-
-- [Build a Workflow](/docs/agent-playground/features/build-workflow): Modify your workflow and add more nodes
-- [Create a Graph](/docs/agent-playground/features/create-graph): Create another graph or manage versions
-- [Versions & Execution](/docs/agent-playground/concepts/versions-and-execution): Understand the execution model in depth
diff --git a/src/pages/docs/agent-playground/guides/build-workflow/configure-an-agent-node.mdx b/src/pages/docs/agent-playground/guides/build-workflow/configure-an-agent-node.mdx
new file mode 100644
index 00000000..35ee7223
--- /dev/null
+++ b/src/pages/docs/agent-playground/guides/build-workflow/configure-an-agent-node.mdx
@@ -0,0 +1,62 @@
+---
+title: "Configure an Agent node"
+description: "Reference another agent's saved version in an Agent node and map your workflow's values onto its inputs."
+---
+
+The Agent Node's configuration form lives in the node drawer, and every field in it sets up which agent this step hands off to. It's where you tell the node which agent to run, which version of that agent, and which of the parent workflow's values feed its inputs. Beyond those three, it carries no configuration of its own.
+
+
+This guide picks up once a workflow with an Agent Node already exists on the canvas; see [Build a workflow](/docs/agent-playground/guides/build-workflow) to add one first.
+
+
+## Open the drawer
+
+Click the Agent Node on the canvas. Its drawer opens with the node's own configuration form.
+
+
+
+*The Agent Node drawer, with an agent and version selected and its Input Mapping rows ready to wire up*
+
+## Choose the agent and version
+
+Under **Agent**, select the agent to nest. The field starts empty, with the placeholder "Select agent".
+
+Under **Version**, select which version of that agent to run. Until an agent is chosen, this field stays on "Select an agent first"; once you choose an agent, its latest non-draft version is preselected here, and you can change it to any of its other versions.
+
+You can select any active or inactive saved version of a **different** agent.
+
+
+The one reference rule Save can still refuse is referencing a second version of an agent you've already referenced elsewhere in this workflow.
+
+
+## Map the inputs
+
+Input Mapping lists one row per input the nested agent expects. Row labels are the nested agent's own input names, set when that agent was built rather than here; they double as the mapping keys. If the nested agent has no inputs, the Input Mapping section doesn't appear at all.
+
+Each row has a **Variable** select, with the placeholder "Select variable", and its options are the output ports of the nodes connected directly into this one, labeled `node_name.output_name`. Pick the upstream output that should feed that input.
+
+For example, say the nested agent expects an `invoice_details` input, and a `classify_invoice` node feeds into this Agent Node. The `invoice_details` row is where you'd pick `classify_invoice.response_1` from the **Variable** select to pass that output down.
+
+
+Leave a row unmapped and no edge is created for it; that input becomes one of the parent workflow's own input variables instead, and per [Set input variables](/docs/agent-playground/guides/build-workflow/set-input-variables), the run refuses to start until it has a value.
+
+
+## Save the node
+
+Click **Save**. A successful save closes the drawer. If the save fails, a toast reads "Failed to save agent node" and the node reverts.
+
+The nested agent's own run appears inside the parent run's results; see [Run an agent](/docs/agent-playground/guides/run-an-agent) for what that looks like.
+
+## Dive deeper
+
+
+
+ The full set of limits and validation rules Agent Playground enforces
+
+
+ Where the parent workflow's own values come from
+
+
+ See where a nested run lands in the parent's results
+
+
diff --git a/src/pages/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node.mdx b/src/pages/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node.mdx
new file mode 100644
index 00000000..e91ac2b5
--- /dev/null
+++ b/src/pages/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node.mdx
@@ -0,0 +1,57 @@
+---
+title: "Configure an LLM Prompt node"
+description: "Name the node, pick its prompt version and model, then save without breaking what's wired into it"
+---
+
+The LLM Prompt node's configuration form lives in the node drawer, and most of it comes from the prompt you pick. Walk the form top to bottom and each choice sets up the next.
+
+
+This guide picks up once a workflow with an LLM Prompt node already exists on the canvas; see [Build a workflow](/docs/agent-playground/guides/build-workflow) to add one first.
+
+
+## Open the drawer
+
+Click the LLM Prompt node on the canvas. Its drawer opens with the node's own configuration form.
+
+## Name the node
+
+**Prompt Name** is required and sits at the top of the form. Typing here sets this node's name: what you type is lowercased, every character other than `a`-`z`, `0`-`9`, and `_` is replaced with an underscore, and leading underscores are stripped.
+
+A name that already belongs to another node on the canvas is refused with "A node with this name already exists". Pick a different name and try again.
+
+## Pick a version
+
+A version select sits beside Prompt Name. Its options are labeled with the version, uppercased. When the selected version hasn't been saved yet, a **Draft** badge appears beside the select, not on the option inside the dropdown.
+
+Not every version can be used here. If the version you pick has an output format the builder doesn't support, the form shows: "This prompt uses an unsupported output format. Only text-based prompts are supported in the agent builder." While that alert shows, the model picker, the **Tools** control, and **Save prompt** are all disabled.
+
+## Choose the model
+
+The model picker selects which LLM this node uses. It's also what unblocks the **Tools** button beside it: Tools stays disabled until a model is chosen, and hovering it before then shows why, "Select a model first".
+
+## Inputs follow the prompt
+
+The node's inputs are generated from the prompt's `{{variable}}` placeholders, and those placeholder names become the node's input names. See [Limits & rules](/docs/agent-playground/reference/limits-and-rules) for naming restrictions. Swapping the prompt text or picking a different version changes that set of placeholders, so it also changes the set of inputs the node exposes.
+
+## Save the prompt
+
+**Save prompt**, at the bottom of the form, writes your changes. If the save fails, a toast reads "Failed to save prompt" and the form stays open so you can retry. On a successful save, the drawer closes.
+
+
+Picking a different version of the prompt can change what this node returns: the output shape downstream nodes expect may shift, since the node's response follows the response format set on the linked version.
+
+
+## Closing with unsaved changes
+
+Close the drawer while a change is unsaved and a dialog titled "Unsaved Changes" asks "You have unsaved changes. Are you sure you want to discard them?" Confirm with **Discard** to drop the edits, or back out of the dialog to go save first.
+
+## Dive deeper
+
+
+
+ Wire the node's inputs to the workflow's values
+
+
+ Try the workflow with the node configured
+
+
diff --git a/src/pages/docs/agent-playground/guides/build-workflow/index.mdx b/src/pages/docs/agent-playground/guides/build-workflow/index.mdx
new file mode 100644
index 00000000..d132b3bc
--- /dev/null
+++ b/src/pages/docs/agent-playground/guides/build-workflow/index.mdx
@@ -0,0 +1,72 @@
+---
+title: "Build a workflow"
+description: "Add nodes to the canvas, wire them up, and save the graph"
+---
+
+This guide picks up once you've [created an agent](/docs/agent-playground/guides/create-agent) and opened it, on an empty canvas. As the running example, say you're building Invoice Triage, a workflow that needs to read each invoice and decide where it goes: an **LLM Prompt** node to classify the invoice, feeding an **Agent Node** that routes it to the right approver. Getting that shape onto the canvas, wiring the two nodes together, and saving the result is what this guide walks through.
+
+
+What you type into either node's own settings isn't covered here; see [Configure an LLM Prompt node](/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node) and [Configure an Agent node](/docs/agent-playground/guides/build-workflow/configure-an-agent-node).
+
+
+## Tour the builder
+
+Open an agent and you land on the **Agent Builder** tab, one of three tabs across the top alongside **Changelog** and **Executions**. Agent Builder holds the canvas itself; the other two sit outside what this guide covers.
+
+The builder splits into three regions. The **node palette** sits on the left, listing the node types you can place. The **canvas** fills the middle and holds the graph as you build it, the nodes and the connections between them. The **node drawer** opens on the right once you select a node, carrying that node's own settings.
+
+
+*Agent Builder, Changelog, and Executions sit across the top; palette, canvas, and drawer make up the three regions underneath*
+
+## Add a node to the canvas
+
+The node palette lists the node types you can add, among them LLM Prompt, described as "Run a prompt against an LLM", and Agent Node, described as "Run an agent through LLM". Get either one onto the canvas by clicking its card, which drops the node straight onto the canvas, or by dragging the card and releasing it wherever you want the node to land.
+
+
+*Click or drag either card onto the canvas*
+
+Reach for an LLM Prompt node for a single, focused call to an LLM, the way Invoice Triage uses one to classify an invoice. Reach for an Agent Node when the step needs to run an agent, the way Invoice Triage uses one to route the invoice to the right approver.
+
+There's a third way to add a node, once you already have one down: a node with no outgoing connection carries a **+** button to its right. Click it and the same node picker opens, so the new node lands already connected to the one before it.
+
+## Connect nodes
+
+A single configured node already runs on its own; wiring is how one node's output becomes the next node's input. Every node has an output handle and an input handle; drag from one node's output handle to the next node's input handle, and the builder draws the connection between them. On Invoice Triage, that means dragging from the classifier's output to the router's input.
+
+
+*Output and input handles sit on the edge of each node; drag from the classifier's output to the router's input to connect them*
+
+Connections follow a few rules:
+- One output can feed as many inputs as you connect it to, so a single node's result can branch into several downstream nodes at once
+- One input takes only one source
+- A connection that loops back into a node's own upstream path draws fine but won't save
+
+Invoice Triage's LLM Prompt node still needs the invoice text itself to classify, and that value comes from outside any node, as an [input variable](/docs/agent-playground/guides/build-workflow/set-input-variables).
+
+## Delete a node
+
+A node can go from two places: a delete icon sits right on the node itself on the canvas, and the node drawer offers the same action for whichever node you have open. The canvas icon removes the node immediately, with no confirmation. The drawer's delete icon opens a dialog titled **Delete Node** that asks "Are you sure you want to delete this node? This action cannot be undone." Click **Delete** to confirm, or close the dialog to keep the node. If the deletion doesn't go through, a "Failed to delete node" toast tells you.
+
+## Save Agent
+
+Click **Save Agent** once the graph looks right. If the graph contains a cycle or a node is left unconfigured, **Save Agent** toasts the error instead of opening anything. Once the graph passes that check, the dialog opens, carrying a **Version** field you can't edit, a **Commit Message** box for a note about the change, and a single action button. That button reads **Save**, or **Save & Run** if you reached the dialog by clicking **Run Agent Workflow** on an unsaved draft, in which case it also runs the agent after saving, the same run covered in [Run an agent](/docs/agent-playground/guides/run-an-agent).
+
+The dialog closes and the graph is saved as a new version; find it later, commit message and all, in the [Changelog tab](/docs/agent-playground/guides/manage-versions).
+
+**Save Agent** itself stays disabled until you have permission to edit the agent, the agent is on a draft, you're on the Agent Builder tab, no run is already in progress, and the canvas has finished loading. Hover a disabled button to see why; without edit permission, the tooltip reads "You don't have permission to edit this agent."
+
+With that saved, Invoice Triage has its shape: an LLM Prompt node feeding an Agent Node, connected and versioned. Configuring what each node actually does comes next.
+
+## Dive deeper
+
+
+
+ The settings behind an LLM Prompt step
+
+
+ The settings behind an Agent Node step
+
+
+ Feed values into the graph from outside any one node
+
+
diff --git a/src/pages/docs/agent-playground/guides/build-workflow/set-input-variables.mdx b/src/pages/docs/agent-playground/guides/build-workflow/set-input-variables.mdx
new file mode 100644
index 00000000..97f13135
--- /dev/null
+++ b/src/pages/docs/agent-playground/guides/build-workflow/set-input-variables.mdx
@@ -0,0 +1,41 @@
+---
+title: "Set input variables"
+description: "Open the Variables drawer, fill in each field, and see what a run does when one's still empty."
+---
+
+A run needs a value for every input variable before it can start. Set them from the builder before you run anything.
+
+
+This guide picks up once a workflow with at least one LLM Prompt node already exists on the canvas; see [Configure an LLM Prompt node](/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node) to add one first.
+
+
+## Open the Variables drawer
+
+In the builder, click **Add input variables** at the top right of the canvas. The drawer opens headed **Variables**, with the subtext "Define values for your prompt variables". The list comes from your workflow's saved version, so a variable you just added won't show up until the node and the agent are saved.
+
+## Fill in each variable
+
+The drawer lists each variable by name, with a field beside it for the value you want this run to use. Give every listed variable a concrete value. If your prompt asks for the invoice text to classify, for example, fill that variable with the actual invoice, such as "Invoice #4521 from Acme Supplies, $2,400 due in 30 days." Only inputs with no incoming connection show up here, since those are [the agent's own inputs](/docs/agent-playground/concepts/understanding-agent-playground#what-isnt-connected-becomes-the-agents-own-input-or-output).
+
+## Save your values
+
+Click **Save** to store your values. If a run is waiting on these variables, the button reads **Save & Run Workflow** instead, and saving starts that run.
+
+
+Close the drawer instead while an edit is unsaved, and a dialog titled "Unsaved Changes" asks "You have unsaved changes. Are you sure you want to close without saving?" Confirm with **Discard Changes** to close without keeping them, or cancel to go back and save first.
+
+
+## A run won't start with an empty variable
+
+Leave any variable blank and a run refuses to start. You'll see the warning "Fill in all variables before running", and the drawer opens on its own with the run held until you fill in what's missing and save. Once every variable has a value, [Run an agent](/docs/agent-playground/guides/run-an-agent) covers starting the run itself.
+
+## Dive deeper
+
+
+
+ Start a run once every variable has a value
+
+
+ What a saved version is and how a run turns it into node records
+
+
diff --git a/src/pages/docs/agent-playground/guides/create-agent.mdx b/src/pages/docs/agent-playground/guides/create-agent.mdx
new file mode 100644
index 00000000..52bbcf57
--- /dev/null
+++ b/src/pages/docs/agent-playground/guides/create-agent.mdx
@@ -0,0 +1,62 @@
+---
+title: "Create an agent"
+description: "Create a new agent from scratch or a template"
+---
+
+Every agent in Agent Playground starts in the same place: a list you open from the sidebar, and a button that drops you onto a blank canvas. This guide gets a new agent onto that canvas and back out again if you ever need to delete it, nothing more.
+
+
+The example built up across this section is an agent called **Invoice Triage**. Create yours under that name so later guides line up with what's on your screen.
+
+
+## Open the list and create an agent
+
+Click **Agents** in the sidebar to see every agent in the workspace. Clicking anywhere on a row opens that agent, so there's no separate open button to look for. Once there are more agents than fit on one page, use the Search box above the table to find one by name, and the pagination controls beneath it to page through the rest. If the workspace has no agents yet, the list is replaced by an empty state instead: a **Create your first agent** heading, the line "Break down complex tasks into sequential steps that build upon each other." underneath, and a **Start creating** button in place of **Create Agent**. It opens the same empty canvas.
+
+Click **Create Agent** in the top right to start a new one. You land straight on an empty builder canvas for a brand-new agent. That agent already exists as an empty draft the moment the canvas opens.
+
+
+If **Create Agent** or, later, **Delete** looks greyed out, hover it: a tooltip explains why, either `You don't have permission to create agents.` or `You don't have permission to delete agents.`
+
+
+## Name your agent
+
+The new agent already has a name at the top of the canvas, auto-generated from the timestamp, like `Agent Aug 11, 2026 4:52 PM`. Click the edit icon next to it, type **Invoice Triage**, and press Enter to save it before you start building.
+
+## Choose how to start
+
+The canvas offers two ways to build the same agent: node by node, or from a template.
+
+### Build node by node
+
+Click **Add first node** to open the [node](/docs/agent-playground/concepts/understanding-agent-playground) picker and start building the canvas yourself.
+
+### Start from a template
+
+Use the **or start from a template** link instead to open a drawer titled **Agent Templates**, with a **Search templates** box at the top for finding one by name. A template is the quicker start when your agent fits a common use case like writing, coding, or research. Pick one and it loads a ready-made agent onto the canvas in place of the empty one, then continue from [Build a workflow](/docs/agent-playground/guides/build-workflow) to keep building on what it gave you.
+
+
+A **Stop** control appears while a template is loading; using it warns that stopping now will erase your progress and restart the setup.
+
+
+## Remove agents you no longer need
+
+Back in the agent list, tick the checkbox on one or more rows. The header above the list swaps to a count of how many you've picked, like 3 Selected, with **Delete** and **Cancel** next to it.
+
+Press **Delete**, and a confirmation dialog titled **Delete agents** asks `Are you sure you want to delete 3 agents?`. Click **Delete** to remove them for good, or **Cancel** to back out and keep them.
+
+Deleting an agent fails when another agent's node still references one of its versions: the error names both the agent you tried to delete and the agent whose node depends on it. Open that referencing agent, find the node pointing to the version you're trying to remove, and repoint or delete that node first. See [Versions & execution](/docs/agent-playground/concepts/versions-and-execution) for more on how those references work.
+
+## Dive deeper
+
+
+
+ Build out the canvas you just created
+
+
+ Run the agent you just created and inspect what each step produced
+
+
+ Save drafts, browse the Changelog, and restore old versions
+
+
diff --git a/src/pages/docs/agent-playground/guides/manage-versions.mdx b/src/pages/docs/agent-playground/guides/manage-versions.mdx
new file mode 100644
index 00000000..4ba42cd2
--- /dev/null
+++ b/src/pages/docs/agent-playground/guides/manage-versions.mdx
@@ -0,0 +1,29 @@
+---
+title: "Manage versions"
+description: "Find a past save by its commit message and preview it on a read-only canvas."
+---
+
+The Changelog tab lets you look back through an agent's version history without disturbing whatever you're currently building, and preview any version on a read-only canvas.
+
+## Open the Changelog tab
+
+From the agent list, open Invoice Triage. It opens on the **Agent Builder** tab, with **Changelog** and **Executions** alongside it across the top; switch to **Changelog** and the view splits into a version list on the left and a canvas preview on the right.
+
+## Read the version list
+
+Each entry in the list carries its version number and the [commit message](/docs/agent-playground/guides/build-workflow) you wrote when you saved it, for example version 6 with the message "Add duplicate invoice check," so you can tell what changed without opening anything. Version numbers only increase with each save, so the entry with the highest number is the newest.
+
+## Preview a version
+
+Click version 6 and its graph renders on the right, exactly as it looked the moment you saved it. This preview is read-only, draft included. Editing only happens on a [draft](/docs/agent-playground/concepts/versions-and-execution), marked with the Draft badge, back in Agent Builder.
+
+## Dive deeper
+
+
+
+ The limits and validation rules Agent Playground enforces
+
+
+ Run a workflow and see what each node produced
+
+
diff --git a/src/pages/docs/agent-playground/guides/run-an-agent.mdx b/src/pages/docs/agent-playground/guides/run-an-agent.mdx
new file mode 100644
index 00000000..0be90e54
--- /dev/null
+++ b/src/pages/docs/agent-playground/guides/run-an-agent.mdx
@@ -0,0 +1,57 @@
+---
+title: "Run an agent"
+description: "What has to be ready before Run Agent Workflow works, and where a run goes once you step away from the builder."
+---
+
+Running a graph in Agent Playground executes every node your edges connect, and gives you a live account of what happened at each one: which node ran, what it received, and what it returned. This guide covers pressing **Run Agent Workflow**, reading a run while it's in progress, and finding it again after you've left the builder.
+
+Running assumes a graph already exists and is saved. This guide runs Invoice Triage, built in [Build your workflow](/docs/agent-playground/guides/build-workflow); build yours first if you haven't. Two more things have to be true before a run starts.
+
+- Every node on the canvas needs to be configured, or the run is refused with "Node not configured" for one node, or "3 nodes are not configured" when more than one is missing setup. [Configure an LLM Prompt node](/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node) or [Configure an Agent node](/docs/agent-playground/guides/build-workflow/configure-an-agent-node) to clear this
+- Every variable the graph needs has to be filled in, or the run is held and the **Variables** drawer opens on its own with "Fill in all variables before running". See [Set input variables](/docs/agent-playground/guides/build-workflow/set-input-variables)
+
+Fix whichever one is blocking you and try again.
+
+## Start the run
+
+Open your graph in the **Agent Builder** tab, then press **Run Agent Workflow** to execute the graph. The label switches to **Rerun Agent Workflow** the next time you run it, so the button itself tells you whether this is a first run or a repeat.
+
+If you've made changes you haven't saved, a dialog titled "Unsaved Changes" stops you before anything runs: "You have unsaved node changes. Running now will use the last saved configuration." Click **Run Anyway** to go ahead with the [last saved version](/docs/agent-playground/concepts/versions-and-execution), or close the dialog and save first if the run needs to reflect your edits.
+
+## Watch it run
+
+Once the run starts, a run panel opens at the bottom of the builder. It lists each node by name, and **Show Outcome** and **Hide Outcome** fold the panel away or bring it back without stopping anything underneath.
+
+While it runs, the node currently executing is the one animating on the canvas. A node marked failed didn't complete, and any node downstream of it is marked skipped rather than running. See [Limits & rules](/docs/agent-playground/reference/limits-and-rules) for the full list of statuses.
+
+Click a node inside the panel, including one marked failed, to see what went into it and what came out. Click the classifier and you'll see the output it handed off, the same value the router receives as its input.
+
+Once the agent finishes, the panel selects the last node that ran.
+
+If the run as a whole can't complete for a reason that isn't pinned to one node, the error you see falls back to a generic "Workflow execution failed". Check the panel's per-node statuses for one marked failed, and click it to see what it received and returned.
+
+## Leave it running
+
+Click **Exit Workflow** while a run is still going and a dialog titled "Leave running workflow?" asks first: "Your workflow will run in the background. You can find it in the Execution tab." Choose **Leave** to step away, or **Cancel** to stay and keep watching.
+
+Exit Workflow doesn't stop the run. A toast confirms it: "Exited workflow. It will continue running in the background." The run keeps executing with the builder closed.
+
+## Find it in Executions
+
+The **Executions** tab sits across the top of the agent alongside **Agent Builder** and **Changelog**; open it to find that run again, or any other run of this graph. A list of runs sits on the left; select one and its per-node detail loads on the right, the same kind of detail the run panel showed while it was in progress. When a step is an [Agent Node](/docs/agent-playground/concepts/understanding-agent-playground#composing-agents-with-an-agent-node), opening it opens the nested run inside it too.
+
+If nothing has run yet, the tab shows "No executions yet", with "Run your workflow from the Agent Builder to see results here" underneath.
+
+## Dive deeper
+
+
+
+ Open the Changelog tab, read the version list, and preview any saved version
+
+
+ The limits and validation rules Agent Playground enforces
+
+
+ Symptom-first fixes for the builder's most common blockers
+
+
diff --git a/src/pages/docs/agent-playground/index.mdx b/src/pages/docs/agent-playground/index.mdx
index 9b2349f5..6e208bff 100644
--- a/src/pages/docs/agent-playground/index.mdx
+++ b/src/pages/docs/agent-playground/index.mdx
@@ -1,77 +1,37 @@
---
-title: "Agent Playground: Visual Workflow Builder"
-description: "Design and run multi-step AI agent workflows on a drag-and-drop canvas. Chain LLM calls, embed sub-agents, version changes, and trace every execution."
+title: "Overview"
+description: "Where to start: what Agent Playground is, and the guides for creating, building, and running an agent."
---
-## About
+## What is Agent Playground?
-Agent Playground is Future AGI's workflow builder for AI agents. It lets you design multi-step AI workflows by dragging nodes onto a canvas and connecting them, no code required.
+Agent Playground is the visual builder under **Agents** in the sidebar, where you assemble a multi-step agent from nodes on a canvas, run it, and read what each step produced. Reach for it once a single prompt can't do the whole job on its own, such as when one step's output needs to feed the next step, or a step needs to call another agent.
-Most AI applications are not a single prompt. They chain steps together: call one model, pass its output to another, combine results, and so on. As these pipelines grow, keeping track of every connection and debugging failures across steps gets harder. When something breaks, tracing which step went wrong means digging through logs.
+The graph you build on the canvas is that agent, one step per node. Every node is one of two types:
-Agent Playground makes this visual. You build your workflow as a graph of connected steps. Each step is a **node**: like an LLM call or a sub-agent. You draw connections between nodes to define how data flows from one step to the next. When you are ready, you hit **Run** and watch each step execute in real time, with results visible per node.
-
-Key capabilities:
-
-- **Visual builder**: drag-and-drop canvas to design workflows without writing code
-- **Real-time execution**: run your workflow and watch each step light up as it completes
-- **Version control**: draft changes safely, activate when ready, roll back if needed
-- **Batch testing**: connect a dataset and run your workflow against hundreds of inputs at once
-- **Reusable components**: embed one workflow inside another for modular, composable designs
-- **Full traceability**: every run is recorded with complete input/output details per step
-
----
-
-## How Agent Playground Connects to Other Features
-
-- **Prompt**: LLM Prompt nodes use prompts you have already built in the Prompt Management system. Update a prompt once, and every workflow using it picks up the change automatically.
-- **Dataset**: Each graph has a linked dataset where you can set up input variables and run experiments. Go to Dataset to add rows, fill in values for each input, and execute your workflow across all of them.
+- **LLM Prompt** runs a prompt you already built in [Prompt Management](/docs/prompt)
+- **Agent Node** runs another saved agent as a single step
---
-## Know the Parts
-
-Before diving in, here is what each term means and how they fit together.
-
-
-
- A **graph** is the container for your entire workflow. Think of it as a project: it has a name, description, team collaborators, and one or more saved versions. You build and run workflows inside a graph.
-
-
- A **node** is one step in your workflow. There are two kinds:
+## Start here
- - **LLM Prompt nodes** call a language model using a prompt template you have set up in Prompt Management. You pick the model, set parameters like temperature, and the node handles the rest.
- - **Agent nodes** embed an entire other workflow as a single step, useful for breaking complex pipelines into reusable building blocks.
-
-
- An **edge** is the line connecting two nodes. It defines how data flows from one step to the next. You create edges by dragging from one node's output to another node's input on the canvas.
-
-
- Versions let you iterate safely. Make changes in a **Draft**, then **Activate** it when you are happy with the result. Previous versions are saved, so you can always go back and pick up from an earlier state.
-
-
- An **execution** is one run of your workflow. It records the status of every step (success, failed, running), how long each took, and what data went in and came out. You can browse past executions to debug issues.
-
-
-
----
-
-## Getting Started
-
-
-
- Learn how graphs, nodes, and connections work together to form workflows.
+
+
+ Create your first agent and manage its versions
-
- Understand the version lifecycle and how workflows run.
+
+ Add nodes, configure them, and connect them on the canvas
-
- Create your first workflow and start building.
+
+ Run an agent and inspect what each step produced
-
- Add steps, configure them, and connect them into a pipeline.
+
+ How graphs, nodes, ports, and edges fit together
-
- Execute workflows, watch results in real time, and browse history.
+
+ How the draft/active version lifecycle and execution model work
+
+Something not working, or need to know a limit before you build? See [Agent Playground FAQ & fixes](/docs/agent-playground/troubleshooting) and [Limits & rules](/docs/agent-playground/reference/limits-and-rules).
diff --git a/src/pages/docs/agent-playground/reference/limits-and-rules.mdx b/src/pages/docs/agent-playground/reference/limits-and-rules.mdx
new file mode 100644
index 00000000..0c60d6a0
--- /dev/null
+++ b/src/pages/docs/agent-playground/reference/limits-and-rules.mdx
@@ -0,0 +1,90 @@
+---
+title: "Limits & rules"
+description: "Character limits, connection rules, and paging behavior across Agent Playground"
+---
+
+These are the limits and validation rules Agent Playground enforces.
+
+## Naming
+
+| Name | Maximum length | Cannot contain |
+|---|---|---|
+| Agent name | 255 characters | No restriction |
+| Node name | 255 characters | `.` `[` `]` `{` `}` |
+| Input name | 100 characters | No restriction |
+| Output name | 100 characters | `.` `[` `]` `{` `}` |
+
+A name that breaks one of these rules can't be saved.
+
+## Connections
+
+Break any of these and the connection won't attach.
+
+| Rule | Behavior |
+|---|---|
+| One source per input | An input accepts exactly one incoming connection; an output can feed any number of inputs |
+| Direction | A connection always runs from an output to an input |
+| Same version | Both connected nodes must belong to the same [version](/docs/agent-playground/concepts/versions-and-execution) |
+| No loops | A connection cannot create a cycle, and a node cannot connect to itself |
+
+## Agent Nodes
+
+See [Agent Node](/docs/agent-playground/concepts/understanding-agent-playground#composing-agents-with-an-agent-node) for what it is.
+
+Breaking any of these means the node can't be saved with that target.
+
+| Rule | Behavior |
+|---|---|
+| Version status | Points only at an active or inactive version of another agent, never a draft |
+| Self-reference | Cannot point at the agent it belongs to |
+| One version per target agent | All Agent Nodes in a version that point at the same target agent must point at the same version of it, for example Agent B v3, not v3 and v4 together |
+| Cycles | Cannot form a cycle of Agent Nodes pointing at each other |
+
+## Versions
+
+Breaking any of these blocks the edit, the activation, or the run.
+
+| Rule | Behavior |
+|---|---|
+| Editable | Only the draft version can be edited |
+| Running version | Exactly one version of an agent runs at a time |
+| Minimum versions | The last remaining version of an agent cannot be removed |
+| Output names | Two outputs left unconnected can't share a name within the same version |
+| Required inputs | Every node's required inputs must be present |
+
+## Runs
+
+### Statuses
+
+| Level | Possible statuses |
+|---|---|
+| Run | pending, running, success, failed, cancelled |
+| Node step | pending, running, success, failed, skipped |
+
+### Execution limits
+
+| Limit | Value |
+|---|---|
+| Concurrent nodes | Up to 10 nodes run at the same time, per run |
+| Step attempts | A step is retried automatically, up to 3 attempts, before it's marked failed |
+| Step timeout | A step is cut off after 1 hour if it hasn't finished |
+
+## Lists
+
+| List | Page size |
+|---|---|
+| All lists | 10 rows per page |
+
+## Keep exploring
+
+
+
+ Nodes, connections, and nested agents, the pieces these rules apply to
+
+
+ How drafts, versions, and runs relate
+
+
+ Run a workflow and read node status from the Executions tab
+
+
diff --git a/src/pages/docs/agent-playground/troubleshooting.mdx b/src/pages/docs/agent-playground/troubleshooting.mdx
new file mode 100644
index 00000000..b48ed5dc
--- /dev/null
+++ b/src/pages/docs/agent-playground/troubleshooting.mdx
@@ -0,0 +1,102 @@
+---
+title: "Agent Playground FAQ & fixes"
+description: "Symptom-first fixes for the builder's most common blockers"
+---
+
+## In this page
+
+Hit a wall in the Agent Builder? Each fix below leads with what you see on screen, so scan for the message or symptom that matches yours. If your problem isn't listed, reach out via [support](https://futureagi.com/contact-us) with the error text and the run or agent ID.
+
+## While building
+
+### The input I'm connecting to already has a source
+
+An input port only accepts one incoming edge. Delete the existing edge into that input before drawing the new one.
+
+See the full set in [Limits & rules](/docs/agent-playground/reference/limits-and-rules#connections).
+
+### The node name I typed gets rejected
+
+`A node with this name already exists`
+
+Every node on the canvas needs a unique name, so pick a different one.
+
+### The prompt shows an inline error in the drawer
+
+`This prompt uses an unsupported output format. Only text-based prompts are supported in the agent builder.`
+
+Pick a text-based version of the prompt; see how to pick a prompt version in [Configure an LLM Prompt node](/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node).
+
+## When saving or running
+
+### Save or Run shows an error toast
+
+`Node not configured`, or `3 nodes are not configured` when several are; the flagged nodes get highlighted on the canvas so you know which ones to open.
+
+Open each flagged node's drawer and finish its form; [Configure an LLM Prompt node](/docs/agent-playground/guides/build-workflow/configure-an-llm-prompt-node) and [Configure an Agent node](/docs/agent-playground/guides/build-workflow/configure-an-agent-node) cover what each form needs.
+
+`Graph contains a cycle. Remove the circular connection before saving.`
+
+The canvas lets you draw a connection that loops the graph back on itself; it's only caught here, when you save or run. Remove the connection that closes the loop and save again.
+
+### Save Agent is greyed out
+
+Several things gate this button, and it stays disabled until all of them clear:
+
+- You're not on the builder tab: switch back from Changelog or Executions
+- You're viewing a saved version, not the draft: see [Manage versions](/docs/agent-playground/guides/manage-versions) for how to get back into a draft
+- A run is still in flight: wait for it to finish
+- The canvas is still loading: wait for it to finish
+- You don't have permission to edit this agent. Hover the button for the tooltip: `You don't have permission to edit this agent.` See [Roles & Permissions](/docs/roles-and-permissions) for who can grant you edit access
+
+### Run Agent Workflow is greyed out
+
+This is gated by the same edit permission as Save Agent. Hover the button for the tooltip: `You don't have permission to run this agent.` See [Roles & Permissions](/docs/roles-and-permissions) for who can grant you edit access.
+
+### The run won't start
+
+`Fill in all variables before running`, or `Failed to validate variables` if the check itself can't complete.
+
+The builder opens the Variables drawer for you and queues the run behind it, so fill in what's missing there; see [Set input variables](/docs/agent-playground/guides/build-workflow/set-input-variables).
+
+### The whole run shows as failed
+
+If the toast reads `Workflow execution failed`, the run never started. Retry it, and if it keeps failing to start, send support the agent ID.
+
+Otherwise, the toast carries the failed node's own error message. Open the run panel, find the failed node, and fix it; see [Run an agent](/docs/agent-playground/guides/run-an-agent).
+
+## During a run
+
+### The run started, but a node's status badge shows skipped
+
+A node is marked skipped, rather than run, when the step feeding it failed. Fix the failed node upstream and rerun; see [Run an agent](/docs/agent-playground/guides/run-an-agent) for how the run panel shows each node's status.
+
+## Managing agents
+
+### I got bounced to the agents list with a missing-agent message
+
+`Missing agent`
+
+The builder shows this and sends you back to the agent list when the URL you opened carries no agent ID at all. Open the agent again from the list instead; see [Create an agent](/docs/agent-playground/guides/create-agent).
+
+### An agent won't delete
+
+An agent stays undeletable while another agent's node still references one of its versions. Open the referencing agent, remove or swap out that node, then delete again; see [Create an agent](/docs/agent-playground/guides/create-agent#remove-agents-you-no-longer-need).
+
+### The last version of an agent won't delete
+
+An agent always keeps at least one version, so its last one can't be removed. Add a new version first if you want to retire the old one; see [Manage versions](/docs/agent-playground/guides/manage-versions).
+
+## Keep exploring
+
+
+
+ Open the agent list, create an agent, and clear out the ones you don't need
+
+
+ What has to be ready before a run starts, and where it goes once you step away
+
+
+ How a draft becomes a saved version, and how a run turns it into node records
+
+
diff --git a/src/pages/docs/annotations/concepts/labels.mdx b/src/pages/docs/annotations/concepts/labels.mdx
new file mode 100644
index 00000000..3e313364
--- /dev/null
+++ b/src/pages/docs/annotations/concepts/labels.mdx
@@ -0,0 +1,66 @@
+---
+title: "Labels"
+description: "Why every score built on a label keeps the same shape and options"
+---
+
+## A reusable question with a fixed answer type
+
+A **label** is a reusable question plus the answer type it accepts. Attach it to a [queue](/docs/annotations/concepts/queues-and-items) and every annotator working that queue answers the same question, the same way, and each answer lands as a [score](/docs/annotations/concepts/scores). The full object model connecting labels, queues, and scores lives in [Understanding Annotation](/docs/annotations/concepts/understanding-annotation); this page is about the label on its own.
+
+## Three rules that follow
+
+**A label's type is locked the moment you create the label** and can't be changed afterward, though you can still edit its name, description, or settings. You can't turn a numeric label into a categorical one, for example: the type fixes both the control the annotator sees and the shape of the value written into every resulting score. Picked the wrong type? [Create a new label](/docs/annotations/guides/create-label) with the right type instead of trying to change this one.
+
+**A label belongs to the org you created it in, not to any single queue, so editing it changes every queue it's attached to.** A queue attaches an existing label rather than copying it, so editing a label's name, description, or settings changes what every attached queue shows and validates against from that point on. Scores already submitted aren't touched; the edit applies to answers submitted after it.
+
+**A queue needs at least one label to exist.** You can't create or save a queue with zero labels attached. The label is what turns a pile of items into something answerable.
+
+ C["Annotator's control · the label's option list"]
+ L --> V["Value shape in every score · the selected option(s)"]
+ L -->|attached to| Q1["Queue · Support quality review · needs 1+ label"]
+ L -->|attached to| Q2["Queue · Onboarding review · needs 1+ label"]`} />
+
+## Five types, five kinds of judgement
+
+- **Categorical** collects one option from a list you define, or several if you allow multiple selection, best for a defect category, a sentiment, or an escalation reason
+- **Numeric** collects a number within the range you set, best for relevance on a 1-to-10 scale or a quality score out of 100
+- **Text** collects free-text feedback in the annotator's own words, best for detail that doesn't reduce to an option or a number
+- **Star Rating** collects a star count on a scale you choose when creating the label, from 1 up to 10 stars, best for a fast overall impression
+- **Thumbs Up/Down** collects a single up or down call, best for a binary pass or fail judgement
+
+The settings each type requires and the exact validation applied to a submitted value live on [Label types & values](/docs/annotations/reference/label-types-and-values).
+
+## Picking a type
+
+Match the type to the shape of the judgement, not the topic:
+
+| If you need | Pick |
+| --- | --- |
+| The answer to be one or more of a few known outcomes | Categorical |
+| The answer to be a quantity | Numeric |
+| The answer explained, not selected | Text |
+| A quick, coarse gut check | Star Rating |
+| The answer to be strictly one of two | Thumbs Up/Down |
+
+If an item needs more than one kind of judgement, attach more than one label to the queue rather than stretching a single label to cover two jobs.
+
+## Why it matters
+
+Standardizing on a small set of labels (one for quality, one for tone) keeps scores comparable across teams. `Response quality`, for instance, is a categorical label with three options, `Good`, `Needs work`, `Wrong`, so everyone answering it is choosing from that exact same set, not inventing their own scale.
+
+## Keep exploring
+
+
+
+ How labels attach to a queue
+
+
+ What a label's answer becomes
+
+
+ Get a usable label into your org
+
+
diff --git a/src/pages/docs/annotations/concepts/queues-and-items.mdx b/src/pages/docs/annotations/concepts/queues-and-items.mdx
new file mode 100644
index 00000000..bd700da1
--- /dev/null
+++ b/src/pages/docs/annotations/concepts/queues-and-items.mdx
@@ -0,0 +1,80 @@
+---
+title: "Queues & Items"
+description: "The operational layer that turns annotation into a managed campaign"
+---
+
+## A queue is the campaign, an item is what's inside it
+
+A **queue** is the campaign. It carries:
+
+- a status
+- the [labels](/docs/annotations/concepts/labels) it's collecting answers for
+- the people working it and the role each one holds
+- **submissions per item**: how many independent annotators must complete an item before it counts as done
+- whether review is required before a submission counts
+
+An **item** is one thing inside that campaign, pulled from one of the sources Annotation covers (see [Understanding Annotation](/docs/annotations/concepts/understanding-annotation)). It lands in the queue when someone adds it there (see [Add items](/docs/annotations/guides/explore-queue/add-items)), moving from untouched to answered as work happens.
+
+A queue moves through four statuses of its own:
+
+- `draft`: doesn't accept annotations yet
+- `active`: the only status that accepts submissions, flips on the moment someone activates it
+- `paused`: temporarily closed to new submissions without marking the queue done
+- `completed`: closed to further submissions, either because a manager marked it done directly or because every item finished on its own; a skipped item blocks that automatic path, so a manager has to close the queue by hand instead
+
+## How a queue and its items relate
+
+An item carries a status of its own, plus a review state layered on top of it when the queue requires review.
+
+The status moves through four values:
+
+- `pending`: waiting to be picked up
+- `in_progress`: being worked on
+- `completed`: done
+- `skipped`: passed over without an answer
+
+On a queue where submissions per item is 1, opening an item reserves it for whoever opened it, whether it's still `pending` or already `in_progress`, so a second annotator can't pick up the same one while it's being worked. If the queue's reservation timeout passes before they act on it, the reservation simply lapses and the item becomes claimable again, the status itself doesn't move. On a queue that needs more than one submission per item, that reservation doesn't apply, so more than one annotator can open the same item at once. See [Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits) for the exact timeout options.
+
+An item is done once the required number of annotators have each completed it, with every label the queue requires scored, unless the queue calls for review. When review is required, a submitted item's review state moves to pending review instead of the item completing outright. A reviewer's approval then completes it, and sending it back moves the status to `in_progress` so the annotator can act on the feedback.
+
+|"has"| QL["Labels attached to the queue"]
+ Q -->|"has"| QA["Annotators: roles annotator, reviewer, manager"]
+ Q -->|"holds"| IT["Item: pending"]
+ IT --> INP["Item: in_progress, reserved"]
+ INP --> D{"Every label the queue requires scored by enough annotators?"}
+ D -->|"no"| INP
+ D -->|"yes, review off"| DONE["Item: completed"]
+ D -->|"yes, review on"| REV["Item: pending review"]
+ REV -->|"approve"| DONE
+ REV -->|"send back"| INP`} />
+
+## Roles decide who can do what
+
+Everyone on a queue holds one or more of three roles. An **annotator** submits [scores](/docs/annotations/concepts/scores). A **reviewer** approves or sends back submissions when the queue requires it. A **manager** configures the queue and its people. Only annotators and managers can actually submit an annotation: holding the reviewer role by itself doesn't grant that.
+
+Two shortcuts save you from adding people by hand. Whoever creates a queue is added to it as a manager the moment it's saved, and admins act as managers automatically without ever being added as members: org admins on every queue in their org, workspace admins on every queue in their workspace.
+
+## The default queue you meet before you build one
+
+You'll often run into a queue before you ever set one up yourself. Future AGI creates at most one default queue per project, dataset, or agent definition, and only the first time that scope actually needs one, not automatically for every one of them, so a project nobody has annotated yet has no default queue. Unlike an ordinary queue, it never auto-completes: because it's meant to keep collecting whatever lands in it indefinitely, no amount of finished items closes it on its own.
+
+## Why it matters
+
+The queue is what makes annotation a managed workflow instead of something one person does off to the side. The item is what keeps that workflow honest one row at a time, holding its own status so a single stuck item never quietly skews the read on how the whole batch is progressing. It's also why the same submissions per item rule works whether that count is one or five: the queue sets the bar, and every item is measured against it independently.
+
+## Keep exploring
+
+
+
+ What an answer becomes once it's submitted
+
+
+ Stand up an active queue with labels and annotators
+
+
+ The tabs, toolbar, and rules inside the queue detail view
+
+
diff --git a/src/pages/docs/annotations/concepts/scores.mdx b/src/pages/docs/annotations/concepts/scores.mdx
index 38abd050..5fc5c9d9 100644
--- a/src/pages/docs/annotations/concepts/scores.mdx
+++ b/src/pages/docs/annotations/concepts/scores.mdx
@@ -1,113 +1,84 @@
---
-title: "Annotation Scores: Unified Data Model"
-description: "Understand the Score model — the unified annotation primitive storing label values, annotator, source type, and queue context across all entity types."
+title: "Scores"
+description: "How a score's identity decides when judgements merge and when they don't"
---
-## About
+## What a score is
-A score is the atomic data record created every time an annotation label is applied to a source entity. It is the unified annotation primitive in FutureAGI, replacing the legacy TraceAnnotation model with a single structure that works identically across traces, spans, sessions, dataset rows, prototype runs, and simulation executions.
+A **score** is one answer, from a person or a system, to one [label](/docs/annotations/concepts/labels) about one [source](/docs/annotations/concepts/understanding-annotation). It's the single record type behind every judgement in Annotation, whether it came from working a [queue item](/docs/annotations/concepts/queues-and-items) directly or from an inline score submitted against a source.
-Every score answers five questions: **what** was annotated (source), **how** it was annotated (label and value), **who** annotated it (annotator), **when** (timestamps), and **why** (optional notes and queue context).
+A score carries the annotator's [answer](/docs/annotations/reference/label-types-and-values) to the label, who or what produced it, and where it came from, and it looks the same regardless of which surface wrote it.
-## Score fields
+## What makes a score unique
-| Field | Type | Description |
-|-------|------|-------------|
-| `id` | UUID | Unique identifier for the score. |
-| `label_id` | UUID | The annotation label that was used. Determines the expected value format. |
-| `value` | JSON | The annotation value. Format varies by label type (string, number, boolean, string array). |
-| `source_type` | string | What kind of entity was annotated. One of the six supported source types. |
-| `source_id` | UUID | The ID of the annotated entity (e.g. trace ID, dataset row ID). |
-| `annotator` | string | Who created the annotation -- a user email or system identifier. |
-| `score_source` | string | Origin of the score: `human` (manual annotation), `model` (LLM-generated), or `auto` (rule-based). |
-| `notes` | string | Optional free-text notes attached to the annotation. Available when **Allow Notes** is enabled on the label. |
-| `queue_item` | UUID | Optional. Links the score to a specific queue item if it was created through the queue workflow. Null for inline annotations. |
-| `created_at` | datetime | When the score was created. |
-| `updated_at` | datetime | When the score was last modified. |
+What decides whether a new judgement becomes its own score or lands on an existing one is a key made of four fields. Change any one of them and you get a different score, not an update to an old one.
-## Source types
+- **Source**: the trace or item the score is about
+- **Label**: the label being answered
+- **Annotator**: the person account that submitted it, whether they worked a queue item or scored it inline
+- **Queue item**: the queue item the score was submitted through, if any
-Scores can target any of the following entity types. The `source_type` and `source_id` fields together form a polymorphic foreign key to the annotated entity.
+This key applies to scores that have an annotator; a score written without one keys differently: see [Scores with no annotator](#scores-with-no-annotator) below. The field names above are conceptual: for their exact names in the API and SDK, see [Data models](/docs/sdk/annotation-queues/data-models).
-| Source Type | Entity | Where it appears |
-|-------------|--------|-----------------|
-| `trace` | An LLM trace from Observe | Trace detail view, LLM Tracing grid |
-| `observation_span` | A specific span within a trace | Span detail within trace tree |
-| `trace_session` | A conversation session (group of traces) | Sessions grid and session detail |
-| `dataset_row` | A row in a dataset | Dataset table view |
-| `call_execution` | A simulation call execution | Simulation results view |
-| `prototype_run` | A prototype/experiment run | Prototype results view |
+Say `priya@yourteam.com` scores a trace's **Response quality** label. Two cases show what the key decides.
-## Two ways to create scores
+**Two queues, two scores.** Priya works the trace's item in the `Support quality review` queue and scores it. That's Score A. The same trace later lands in a second queue, `Escalation audit`, and she scores it again with the same label. That's Score B. Source, label, and annotator match between A and B, but the queue item doesn't: `Support quality review` on A, `Escalation audit` on B. Two submissions, two independent scores, neither overwrites the other. The same person's judgement through two different queues is two separate pieces of evidence, not a correction.
-### Queue workflow (managed)
+**Two writes, one queue item.** Now say Priya instead scores that same trace on Response quality inline, then does it again the same way. Both writes resolve to the same queue item, the source's default one, so all four fields match this time: source, label, annotator, and queue item are identical between the two writes. That's Score C: the second write lands on it and updates it, instead of creating a new one.
-Scores created through an annotation queue are linked to a queue item via the `queue_item` field. The queue manages assignment, progress tracking, and completion logic.
+ SC1["Score A"]
+ SRC --> SC2["Score B"]
+ SRC --> SC3["Score C"]
+ LBL --> SC1
+ LBL --> SC2
+ LBL --> SC3
+ ANN --> SC1
+ ANN --> SC2
+ ANN --> SC3
+ QI1 --> SC1
+ QI2 --> SC2
+ QI3 --> SC3
+ W2["Second inline write · same key"] -->|merges, old value kept in history| SC3`} />
-
-
- Select entities (traces, dataset rows, etc.) in their respective views and click **Add to Queue**. Each becomes a queue item.
-
-
- Click **Start Annotating** on the queue detail page. The workspace presents items one at a time with the queue's labels. Each submitted annotation creates a score.
-
-
- When all labels are scored by the required number of annotators, the queue item auto-completes.
-
-
+### Scores with no annotator
-### Inline annotation (direct)
+A score written without an annotator uses a narrower key: source and label alone, with no queue item and no annotator in it. That key guarantees at most one such score can exist for a given source and label: see [What a score remembers](#what-a-score-remembers) below for what happens when its value changes.
-Scores can also be created directly from the detail view of any supported entity -- without going through a queue. Inline annotations have `queue_item` set to null.
+## What a score remembers
-- **Trace detail**: Open a trace, expand the annotation panel, and apply any label from your organization.
-- **Session grid**: Annotation columns appear directly in the sessions table for quick scoring.
-- **Dataset view**: Annotate individual rows from the dataset table.
+Editing a score's value doesn't erase what was there before. The prior value is appended to the score's history before the new one is written, so the current answer and the trail of what it used to be both live on the same record, visible in the queue's annotation panel.
-Inline annotations are ideal for ad-hoc feedback during investigation or review. They produce the same Score records and appear alongside queue-created scores in all views and exports.
+Every score also carries a `score_source` tag recording where it came from: `human` or `api`. It's a record of origin you can read back, not a setting you choose.
-## Value formats by label type
+## A score is its own record
-The `value` field in a score is JSON. Its shape depends on the label type:
+A score doesn't depend on the queue item it came from. The queue item it names is provenance, a record of which pass through which queue produced it, not something the score needs to keep existing. That's why an inline score and a queue score show up side by side in the same list: they're the same kind of record, and the queue item is optional context on either one.
-| Label Type | Value Example | JSON Type |
-|------------|--------------|-----------|
-| Categorical (single) | `"Positive"` | string |
-| Categorical (multi) | `["Relevant", "Accurate"]` | string array |
-| Numeric | `7` | number |
-| Text | `"Consider rephrasing the second paragraph."` | string |
-| Star Rating | `4` | number |
-| Thumbs Up/Down | `true` | boolean |
+## Why it matters
-## Where scores appear
+- Scoring the same source in more than one queue never collides, so a support audit and a compliance audit can run over the same traces independently
+- Correcting a score is safe: the old value doesn't disappear, it's superseded on the same record
+- Every view of a source, and every export, shows one list of scores no matter which surface wrote them
-Scores are surfaced everywhere the annotated entity is displayed:
+## Keep exploring
-- **Trace detail view** -- Annotation panel shows all scores for the trace and its spans.
-- **Sessions grid** -- Dynamic annotation columns display score values inline with session data. Filter and sort by annotation values.
-- **Dataset table** -- Score values appear as columns alongside dataset row data.
-- **Queue detail** -- The items tab shows all scores submitted for each queue item.
-- **API** -- Query scores programmatically with filters on source type, label, annotator, and date range.
-
-## Bidirectional sync
-
-Scores created through different paths stay synchronized:
-
-- An annotation submitted on a trace via Observe **automatically** creates a corresponding score visible in any queue containing that trace.
-- A score submitted through a queue workflow is **immediately** visible in the trace detail and session grid views.
-
-This ensures a single source of truth regardless of where the annotation originated.
-
-## Next Steps
-
-
-
- Learn about the five label types that define score value formats.
+
+
+ Work a queue and watch each submission become a score
-
- Understand how queues manage the annotation lifecycle and produce scores.
+
+ Score a source inline, no queue involved
-
- Walk through the full flow from label creation to submitted scores.
+
+ Pull scores out as a dataset or a file
diff --git a/src/pages/docs/annotations/concepts/understanding-annotation.mdx b/src/pages/docs/annotations/concepts/understanding-annotation.mdx
new file mode 100644
index 00000000..07553513
--- /dev/null
+++ b/src/pages/docs/annotations/concepts/understanding-annotation.mdx
@@ -0,0 +1,67 @@
+---
+title: "Understanding Annotation"
+description: "One shared model so every judgement, however it's made, lands as a comparable score"
+---
+
+## Label, queue, item, score
+
+**Annotation** is how a person turns their judgement on a piece of AI output (a rating, a category, a correction) into a record Future AGI can compare against every other judgement made the same way. An annotator, someone on your team, works through a queue, or judges a source directly. Either path ends at the same four objects: a label, a queue, an item, and a score.
+
+A [label](/docs/annotations/concepts/labels) is a reusable question with a fixed answer type: text, a number, a category, a star rating, thumbs up or down. Define it once and reuse it wherever you want that same question answered.
+
+A [queue](/docs/annotations/concepts/queues-and-items) attaches one or more labels and holds the items waiting to be judged against them. A queue can't exist with zero labels; the labels are what an annotator sees when they open an item.
+
+Each item a queue holds points at exactly one **source**:
+
+- A [trace](/docs/observe/concepts/traces)
+- A [span](/docs/observe/concepts/spans)
+- A [session](/docs/observe/concepts/sessions)
+- A Simulation (a simulated voice or text call)
+- A prototype run (a prompt run over a dataset)
+- A dataset row
+
+There's no such thing as an item pointing at two sources, or none.
+
+Every answer to a label, whether it came from working an item or from judging a source directly, becomes one [score](/docs/annotations/concepts/scores).
+
+## How the four pieces fit
+
+|attached to| Q["Queue · one or more labels"]
+ Q -->|holds| I["Item"]
+ SRC{{"Source · exactly one of: trace, span, session, Simulation, prototype run, dataset row"}}
+ I -->|is about| SRC
+ I -->|produces| SC["Score"]
+ SC -->|points at| SRC
+ L -->|shapes the answer on| SC`} />
+
+### One trace, judged two ways
+
+Take a trace from the `support-agent` project. Route it into the `Support quality review` queue, and it becomes an item whose one source is that trace. An annotator opens the item, answers the queue's `Response quality` label, and that answer lands as a score referencing the trace, the label, and the item.
+
+The same trace is judged directly with no queue picked, in the trace's own view in Observe, and someone answers `Response quality` on it there. That judgement doesn't skip the item: the inline judgement lands as the same kind of score record, pointing at the trace and the label. Both show up wherever the trace's judgements are read back.
+
+## Why it matters
+
+- An item points at exactly one source, so anything reading the score back never has to guess which of six possible sources it was about
+- A judgement made inline is never a lesser record: it resolves to a queue item just like a queue-worked one, so filtering, exporting, or displaying scores never has to special-case where they came from
+- A label defined once and attached wherever it's needed means the same question, `Response quality` in a support queue and in a compliance queue, produces answers that land in one comparable set of scores
+
+## Keep exploring
+
+
+
+ Answer types, from a star rating to free text
+
+
+ Roles, statuses, and how an item gets to complete
+
+
+ The uniqueness grain behind every judgement
+
+
+ Attach labels, add items, and activate it for annotators
+
+
diff --git a/src/pages/docs/annotations/features/add-items.mdx b/src/pages/docs/annotations/features/add-items.mdx
deleted file mode 100644
index a779842e..00000000
--- a/src/pages/docs/annotations/features/add-items.mdx
+++ /dev/null
@@ -1,91 +0,0 @@
----
-title: "Add Items to Annotation Queues"
-description: "Add traces, spans, sessions, dataset rows, prototype runs, and simulation calls to annotation queues for structured human review in Future AGI."
----
-
-## About
-
-Items are the bridge between your data and your annotation workflow. Each item links a source -- a trace, span, session, dataset row, prototype run, or simulation call -- to a queue. When you add items to a queue, annotators can review the source content and apply the queue's labels.
-
-## Supported source types
-
-| Source Type | Where to find | Description |
-|-------------|--------------|-------------|
-| Trace | Observe > Traces | Full LLM trace with input, output, metadata, latency, tokens, and cost |
-| Observation Span | Observe > Trace detail > specific span | An individual span within a trace |
-| Session | Observe > Sessions | A conversation session (group of related traces) |
-| Dataset Row | Datasets > select dataset | An individual row in a dataset |
-| Prototype Run | Prototype > execution history | A prototype experiment run |
-| Simulation Call | Simulation > call logs | A simulated voice or text call execution |
-
-## How to add items from Observe
-
-
-
- Go to your **Observe** project and open the **Traces** view.
-
-
-
- Use the checkboxes to select one or more traces you want to annotate.
-
-
-
- Click the **Add to Queue** button in the toolbar. A dialog opens where you can search for and select the target queue.
-
-
-
- Select the queue and click **Add**. The selected traces appear as items in the queue's **Items** tab with a **Pending** status.
-
-
-
-## How to add from other sources
-
-The flow is the same across all source types:
-
-- **Datasets** -- Navigate to a dataset, select rows using checkboxes, and click **Add to Queue**.
-- **Sessions** -- Open Observe > Sessions, select sessions, and click **Add to Queue**.
-- **Prototyping** -- Open a prototype's execution history, select runs, and click **Add to Queue**.
-- **Simulation** -- Open simulation call logs, select calls, and click **Add to Queue**.
-
-## Managing items in a queue
-
-Open a queue's detail page and go to the **Items** tab to see all items and their statuses.
-
-
-
-### Filtering items
-
-- **By status** -- Filter by Pending, In Progress, Completed, or Skipped.
-- **By source type** -- Show only items from a specific source (e.g., traces only).
-- **My Items** -- Toggle to see only items assigned to you.
-
-### Removing items
-
-- Select one or more items using checkboxes and click **Remove Selected**.
-- Or click the `x` button on an individual row to remove a single item.
-
-### Bulk operations
-
-Use the select-all checkbox to select all visible items, then apply bulk actions like remove.
-
-
-Duplicate items are silently skipped. If a source is already in the queue, adding it again has no effect. The response shows how many items were added versus how many were duplicates.
-
-
-
-For large-scale annotation campaigns, use the SDK to programmatically add items to queues. See the [Python SDK](/docs/annotations/sdk/python) or [JavaScript SDK](/docs/annotations/sdk/javascript) guide.
-
-
-## Next steps
-
-
-
- Start annotating items in the annotation workspace.
-
-
- Understand how queue items flow through statuses and assignment.
-
-
- Add items programmatically via the REST API.
-
-
diff --git a/src/pages/docs/annotations/features/analytics.mdx b/src/pages/docs/annotations/features/analytics.mdx
deleted file mode 100644
index 350f29ec..00000000
--- a/src/pages/docs/annotations/features/analytics.mdx
+++ /dev/null
@@ -1,96 +0,0 @@
----
-title: "Annotation Analytics & IAA Metrics"
-description: "Track queue progress, annotator throughput, label distribution, and inter-annotator agreement using Cohen's and Fleiss' Kappa metrics."
----
-
-## About
-
-Every annotation queue includes a built-in analytics dashboard that shows progress, throughput, and quality metrics. Use it to monitor how your annotation campaign is going and to identify issues before they compound.
-
-## Accessing analytics
-
-Open a queue and click the **Analytics** tab.
-
-
-
-## Overview stats
-
-The top of the analytics view shows four key numbers at a glance:
-
-- **Total items** -- Number of items currently in the queue.
-- **Completed** -- Number of items that have been fully annotated.
-- **Completion rate** -- Percentage of items completed out of the total.
-- **Average completions per day** -- Rolling daily throughput across the queue's lifetime.
-
-## Status breakdown
-
-A visual bar displays the distribution of item statuses:
-
-- **Completed** (green) -- All required annotations collected.
-- **In Progress** (blue) -- At least one annotation submitted, more required.
-- **Pending** (gray) -- No annotations yet.
-- **Skipped** (orange) -- Annotator chose to skip the item.
-
-## Daily throughput chart
-
-A bar chart showing the number of completions over the last 30 days. Use it to spot trends, identify slowdowns, and measure annotator velocity over time.
-
-## Annotator performance table
-
-| Column | Description |
-|--------|-------------|
-| Annotator | Name and email of the team member |
-| Completed | Number of items this annotator has completed |
-| Last Active | Timestamp of their most recent annotation |
-
-## Label distribution
-
-For each label attached to the queue, the analytics view shows the frequency of each value:
-
-- **Categorical** -- Option counts (e.g., "Positive: 45, Negative: 23, Neutral: 12").
-- **Numeric / Star** -- Distribution histogram across the value range.
-- **Thumbs** -- Up vs. down counts.
-- **Text** -- Total annotation count (text values are not aggregated).
-
-## Inter-Annotator Agreement
-
-Switch to the **Agreement** tab to see consistency metrics between annotators scoring the same items.
-
-**Metrics used:**
-
-- **Cohen's Kappa** -- Used when exactly 2 annotators have scored the same items.
-- **Fleiss' Kappa** -- Used when 3 or more annotators have scored the same items.
-
-The view shows a per-label agreement breakdown so you can pinpoint which labels have the most disagreement.
-
-**Interpreting Kappa values:**
-
-| Kappa Value | Interpretation |
-|-------------|---------------|
-| < 0.20 | Poor |
-| 0.21 -- 0.40 | Fair |
-| 0.41 -- 0.60 | Moderate |
-| 0.61 -- 0.80 | Substantial |
-| 0.81 -- 1.00 | Almost perfect |
-
-
-Agreement metrics require `annotations_required` to be set to 2 or more in your queue settings, and at least 2 annotators must have scored the same items for results to appear.
-
-
-
-If agreement is low, review your annotation instructions and consider adding clearer guidelines or simplifying label options. Small improvements to instructions often produce large jumps in agreement.
-
-
-## Next steps
-
-
-
- Export completed annotations as datasets for fine-tuning or evaluation.
-
-
- Learn the annotation workspace and keyboard shortcuts.
-
-
- Understand queue architecture, assignment modes, and lifecycle.
-
-
diff --git a/src/pages/docs/annotations/features/annotate.mdx b/src/pages/docs/annotations/features/annotate.mdx
deleted file mode 100644
index 0922f2dd..00000000
--- a/src/pages/docs/annotations/features/annotate.mdx
+++ /dev/null
@@ -1,107 +0,0 @@
----
-title: "Annotate Items in the Workspace"
-description: "Use the annotation workspace to label traces, sessions, and datasets with categorical, numeric, star, and thumbs inputs plus keyboard shortcuts."
----
-
-## About
-
-The annotation workspace is where annotators provide feedback on queue items. It presents the source content alongside the queue's labels in a focused, distraction-free view designed for fast, consistent annotation.
-
-## How to start annotating
-
-
-
- Navigate to a queue and click the **Start Annotating** button. You can also go to the queue's **Items** tab and click on any individual item.
-
- The workspace opens in a dedicated view.
-
- 
-
-
-
- The **left panel** (~60% of the screen) displays the source content. What you see depends on the source type:
-
- | Source Type | What is displayed |
- |-------------|-------------------|
- | Trace | Full trace tree with expandable spans -- input, output, metadata, latency, tokens, cost |
- | Dataset Row | All fields and values from the dataset row |
- | Session | Conversation history with expandable individual traces |
- | Prototype Run | Prompt, response, and model information |
- | Simulation Call | Transcript, analytics, and audio player (for voice calls) |
-
-
-
- The **right panel** (~40% of the screen) shows each label as a section with a colored header. Fill in values based on the label type:
-
- - **Categorical** -- Click a radio button (single-choice) or checkbox (multi-choice). Use number keys `1`--`9` for quick selection.
- - **Numeric** -- Drag the slider or type directly in the input field. Values are enforced within the configured min/max bounds.
- - **Star** -- Click a star to set the rating. Use number keys `1`--`N` where N is the number of stars.
- - **Thumbs Up/Down** -- Click the **Yes** or **No** button. Use key `1` for thumbs up or `2` for thumbs down.
- - **Text** -- Type in the text area. A character count is shown. Input is saved with a 300ms debounce.
-
-
-
- If the queue's labels have **Allow Notes** enabled, an optional free-text field appears at the bottom of the labels panel. Use it to add context or comments about your annotation.
-
-
-
- Click **Submit & Next** or press `Ctrl+Enter` (`Cmd+Enter` on Mac) to save your annotations and advance to the next item.
-
- - An item is marked as **Completed** when all required labels have been scored.
- - If the queue requires multiple annotators, the item stays **In Progress** until the required number of annotators have submitted.
-
-
-
-## Keyboard shortcuts
-
-Use keyboard shortcuts for significantly faster annotation speed.
-
-| Shortcut | Action |
-|----------|--------|
-| `Tab` / `Shift+Tab` | Navigate between labels |
-| `1`--`9` | Select a categorical option or set a star rating |
-| `Ctrl+Enter` / `Cmd+Enter` | Submit and move to next item |
-| `S` | Skip current item |
-| `←` / `→` | Previous / next item |
-| `?` | Toggle keyboard shortcuts help |
-
-
-Keyboard shortcuts can increase annotation speed by 3--5x. Press `?` in the workspace at any time to see all available shortcuts.
-
-
-## Instructions panel
-
-If the queue creator wrote annotation instructions, they appear in a collapsible section above the labels. Instructions are rendered as markdown and typically include criteria, examples, and edge-case guidance. Review them before starting your first annotation.
-
-## Skipping items
-
-Click the **Skip** button in the header or press `S` to skip the current item and move to the next one. Skipped items can be revisited later -- they are not marked as completed.
-
-## Navigation
-
-- Use the **Previous** and **Next** buttons in the footer to move between items.
-- A position indicator shows your current item (e.g., "5 of 50").
-- The workspace maintains a history of up to 50 visited items for easy back-navigation.
-- A progress bar in the header shows overall completion (X of Y completed).
-
-## Completion
-
-When all items in the queue have been annotated, a success screen appears with completion statistics.
-
-
-If an item's source has been deleted, the workspace displays a "Source item has been deleted" message. If another annotator has reserved the item, a lock icon is shown and you will be routed to the next available item.
-
-
-## Next steps
-
-
-
- Annotate directly from trace detail, session grid, or dataset views without opening a queue.
-
-
- Export annotated data as training or evaluation datasets.
-
-
- View completion rates, annotator activity, and label distributions.
-
-
diff --git a/src/pages/docs/annotations/features/automation.mdx b/src/pages/docs/annotations/features/automation.mdx
deleted file mode 100644
index dfe8c83d..00000000
--- a/src/pages/docs/annotations/features/automation.mdx
+++ /dev/null
@@ -1,73 +0,0 @@
----
-title: "Annotation Automation Rules"
-description: "Create condition-based rules to automatically add matching traces, spans, or sessions to annotation queues without manual curation."
----
-
-## About
-
-Automation rules let you define conditions that automatically trigger actions on queue items -- such as auto-adding items that match certain criteria or pre-filling label values based on span attributes. Instead of manually curating queue contents, you set the rules once and let matching items flow in automatically.
-
-## How to set up an automation rule
-
-
-
- Open a queue and go to the **Rules** tab.
-
-
-
- Click the **Create Rule** button.
-
-
-
- Fill in the rule configuration:
-
- - **Name** -- A descriptive rule name so your team knows what it does at a glance.
- - **Source Type** -- Which type of items this rule applies to (e.g., traces, spans).
- - **Conditions** -- Define match criteria:
- - **Field** -- The attribute to evaluate (e.g., span attribute, metric name).
- - **Operator** -- The comparison operator (equals, greater than, less than, contains).
- - **Value** -- The threshold or match string.
- - **Enabled** -- Toggle the rule on or off.
-
-
-
- Click **Save**. The rule is now active and will be evaluated when new items are added to the queue.
-
-
-
-## Preview and evaluate
-
-Before enabling a rule in production, use these tools to validate it:
-
-- **Preview** -- Click the **Preview** button to see which existing queue items would match the conditions without actually triggering any actions.
-- **Evaluate** -- The **Evaluate** action tests the rule against current items and shows detailed match results, so you can fine-tune conditions before going live.
-
-## Example rules
-
-| Rule Name | Condition | Action |
-|-----------|-----------|--------|
-| Flag low scores | eval_score < 0.5 | Auto-add to review queue |
-| Long responses | token_count > 1000 | Auto-add for quality check |
-| Error traces | status = "error" | Auto-add for analysis |
-
-
-Automation rules are evaluated when new items are added to the queue. Existing items can be tested using the Evaluate action but are not retroactively processed unless you trigger evaluation manually.
-
-
-
-Automation rules are a powerful feature still being expanded. Check back for new condition types and actions as they become available.
-
-
-## Next steps
-
-
-
- Set up the queues that your automation rules feed into.
-
-
- Learn about manual and programmatic ways to add items alongside automation.
-
-
- Monitor the items your rules are adding and track annotation progress.
-
-
diff --git a/src/pages/docs/annotations/features/export.mdx b/src/pages/docs/annotations/features/export.mdx
deleted file mode 100644
index 373276d0..00000000
--- a/src/pages/docs/annotations/features/export.mdx
+++ /dev/null
@@ -1,86 +0,0 @@
----
-title: "Export Annotations to Dataset or File"
-description: "Export completed annotation queue results to a Future AGI dataset or download as JSON/CSV for fine-tuning, evaluation, and offline analysis."
----
-
-## About
-
-Export lets you turn annotation results from a queue into a structured dataset you can use for fine-tuning, evaluation, or offline analysis. You can export directly into a FutureAGI dataset or download as JSON/CSV.
-
-## Export to Dataset
-
-
-
- Open queue detail and click the **Export to Dataset** button in the header.
-
-
-
- Create a **new dataset** by entering a name, or select an **existing dataset** from the dropdown.
-
-
-
- Optionally filter by item status. By default, only completed items are included.
-
-
-
- Click **Export**. The annotations are written as rows in the target dataset with all label values as columns.
-
-
-
-## Export as JSON/CSV
-
-
-
- Open queue detail and click the **Export** button. Choose your format -- **JSON** or **CSV**.
-
-
-
- Optionally filter by item status to include only the records you need.
-
-
-
- Click **Download**. The file is generated and saved to your local machine.
-
-
-
-## Export data structure
-
-Each exported record contains the following fields:
-
-| Field | Description |
-|-------|-------------|
-| item_id | Queue item ID |
-| source_type | Type of annotated source (trace, span, session, etc.) |
-| source_id | ID of the annotated entity |
-| status | Item status (completed, skipped, etc.) |
-| annotations | Array of label values with annotator info |
-| notes | Annotator notes (if any) |
-
-## When to use exported data
-
-- **Fine-tuning** -- Use annotated traces as training data for model improvement.
-- **Evaluation datasets** -- Create golden datasets for automated eval pipelines.
-- **Quality reports** -- Analyze annotation patterns and model failure modes offline.
-- **Model comparison** -- Compare model outputs across annotated dimensions.
-
-
-Export to Dataset creates a full FutureAGI dataset that you can use with all dataset features including experiments, evaluations, and prompt management.
-
-
-
-For programmatic export, use the [Queues API](/docs/api/annotations/queues/export) or the [SDK export methods](/docs/annotations/sdk/python).
-
-
-## Next steps
-
-
-
- Review annotation progress and agreement before exporting.
-
-
- Learn about FutureAGI datasets and what you can do with exported data.
-
-
- Export annotations programmatically via the REST API.
-
-
diff --git a/src/pages/docs/annotations/features/inline.mdx b/src/pages/docs/annotations/features/inline.mdx
deleted file mode 100644
index a64920ec..00000000
--- a/src/pages/docs/annotations/features/inline.mdx
+++ /dev/null
@@ -1,88 +0,0 @@
----
-title: "Inline Annotations: Ad-Hoc Feedback"
-description: "Score traces, sessions, dataset rows, and prototype runs directly from their detail views without setting up an annotation queue."
----
-
-## About
-
-Inline annotations let you score any trace, session, or prototype execution directly from its detail view -- no queue setup required. The InlineAnnotator component appears in the right sidebar of every detail drawer, so you can leave feedback the moment you spot something interesting.
-
-Best for one-off feedback, quick quality checks, or ad-hoc labeling during debugging.
-
-## How to annotate inline from Observe
-
-
-
- Go to your Observe project and click any trace to open the detail drawer.
-
-
-
- Click the **Annotations** tab in the right panel.
-
-
-
- Click the **Annotate** button to enter edit mode.
-
-
-
- Select labels and provide values. The input types are the same as queue-based annotation -- categorical, numeric, text, star, or thumbs up/down.
-
-
-
- Optionally add free-text notes to provide extra context for your annotation.
-
-
-
- Click **Save** to store your annotations. They are immediately visible to your team and available via the API.
-
-
-
-## From Sessions
-
-Same flow -- open a session, switch to the **Annotations** tab, and click **Annotate**. Session-level annotations are tracked separately from individual trace annotations within the session.
-
-## From Prototyping
-
-Open a prototype execution, then click into the trace detail drawer. The **Annotations** tab is available in the right panel -- click **Annotate** to score the execution.
-
-## From Simulation Call Logs
-
-Open a call log detail. The **Annotations** tab appears in the right section of the detail view -- click **Annotate** to score the call.
-
-## Adding new labels inline
-
-You can create labels without leaving the annotation sidebar:
-
-- Click the **Add Label** button in the annotation sidebar.
-- Create a new label or select from your existing labels.
-- The label immediately appears in the annotation form, ready to use.
-
-## Inline vs Queue-based
-
-| Feature | Inline | Queue-based |
-|---------|--------|-------------|
-| Best for | Quick one-off annotations | Structured campaigns |
-| Setup required | None | Create queue, add items |
-| Assignment | Self-serve | Manual, Round Robin, Load Balanced |
-| Progress tracking | Per-score only | Full queue progress + analytics |
-| Multi-annotator | Manual coordination | Built-in agreement metrics |
-| Export | Individual scores | Bulk export to dataset |
-| Keyboard shortcuts | No | Yes (full shortcut support) |
-
-
-Use inline annotations for quick feedback during debugging. Switch to queues when you need structured annotation campaigns with progress tracking and inter-annotator agreement.
-
-
-## Next steps
-
-
-
- Create the labels you'll use for inline annotation.
-
-
- Set up queues for structured annotation campaigns.
-
-
- Understand how scores unify inline and queue-based annotations.
-
-
diff --git a/src/pages/docs/annotations/features/labels.mdx b/src/pages/docs/annotations/features/labels.mdx
deleted file mode 100644
index fe710e97..00000000
--- a/src/pages/docs/annotations/features/labels.mdx
+++ /dev/null
@@ -1,132 +0,0 @@
----
-title: "Annotation Labels: 5 Types Explained"
-description: "Create and configure annotation labels: categorical, numeric, text, star rating, and thumbs up/down. Reusable across all queues in your organization."
----
-
-## About
-
-An annotation label is a reusable template that defines what feedback annotators provide. Labels are organization-scoped: once created, any queue in your workspace can use them. This keeps annotation criteria consistent across teams and projects.
-
-Each label has a type that determines the UI control annotators see and the value format stored in the resulting score.
-
----
-
-## Label Types
-
-| Type | Description | Example Use Case | Value Format |
-|------|-------------|------------------|--------------|
-| **Categorical** | Predefined list of options. Supports single-choice or multi-choice. Can be used for auto-annotation. | Sentiment analysis: Positive, Negative, Neutral | `string` (single) or `string[]` (multi) |
-| **Numeric** | A number within a defined range. | Relevance score from 1 to 10 | `number` |
-| **Text** | Free-form text input for open-ended feedback. | Grammar corrections or rewrite suggestions | `string` |
-| **Star Rating** | Visual star selector for quick quality ratings. | Overall response quality | `number` (1 to N) |
-| **Thumbs Up/Down** | Binary pass/fail toggle. The fastest annotation type. | Helpfulness check: was this answer useful? | `boolean` |
-
-### Which type should I use?
-
-| Scenario | Recommended Type | Why |
-|----------|-----------------|-----|
-| Classify responses into fixed categories | **Categorical** | Predefined options ensure consistency and enable aggregation |
-| Rate quality on a fine-grained scale | **Numeric** | Continuous range captures nuance that categories miss |
-| Collect corrections, rewrites, or explanations | **Text** | Free-form input gives annotators maximum flexibility |
-| Quick quality gut-check (1-5 stars) | **Star Rating** | Visual stars are fast and intuitive for subjective quality |
-| Binary accept/reject decisions | **Thumbs Up/Down** | Fastest annotation type: one click per item |
-| Multiple dimensions per item | Combine multiple labels in one queue | Attach several labels to a single queue for multi-dimensional annotation |
-
-### UI appearance by type
-
-| Type | Annotator UI |
-|------|-------------|
-| Categorical (single) | Radio buttons for each option |
-| Categorical (multi) | Checkboxes for each option |
-| Numeric | Number input with stepper or slider |
-| Text | Multi-line text area |
-| Star Rating | Clickable star icons |
-| Thumbs Up/Down | Thumb up and thumb down buttons |
-
----
-
-## Creating a Label
-
-
-
- Go to **Annotations** in the left sidebar, then open the **Labels** tab. Click **Create Label**.
-
- 
-
-
-
- Fill in the **Name** field (required) and an optional **Description** to help annotators understand the label's purpose.
-
-
-
- Choose the label **Type**. This cannot be changed after creation, so choose carefully.
-
-
-
- Each type has its own configuration options:
-
- - **Categorical**: Add at least two options. Toggle **Allow multiple selection** if annotators should be able to pick more than one option.
- - **Numeric**: Set **Min**, **Max**, and **Step size** values. Choose the display format: **Slider** or **Buttons**.
- - **Text**: Set **Placeholder text**, **Min character length**, and **Max character length**.
- - **Star**: Set the **Number of stars** (1-10, default 5).
- - **Thumbs Up/Down**: No additional settings needed.
-
- 
-
-
-
- Toggle **Allow Notes** if you want annotators to add free-text commentary alongside their label value. Notes are stored in the `notes` field of the resulting score and are available in exports and the API.
-
-
-
- Click **Save**. The label is now available for use in any queue.
-
-
-
----
-
-## Managing Labels
-
-| Action | How |
-|---|---|
-| Edit | Click a label row or use the menu and select **Edit**. You can change the name, description, and type-specific settings, but the type itself is immutable. |
-| Duplicate | Use the menu and select **Duplicate**. Creates a copy you can customize. |
-| Archive | Use the menu and select **Archive**. Soft-deletes the label. Archived labels can be restored. |
-| Search | Use the search bar at the top to filter labels by name. |
-| Filter by type | Use the type dropdown to show only labels of a specific type. |
-
----
-
-## Label Type Settings Reference
-
-| Type | Settings | Default |
-|------|----------|---------|
-| Categorical | `options` (list), `multi_choice` (bool) | `multi_choice`: false |
-| Numeric | `min`, `max`, `step_size` | 0, 10, 1 |
-| Text | `placeholder`, `min_length`, `max_length` | empty string, 0, 5000 |
-| Star | `no_of_stars` | 5 |
-| Thumbs Up/Down | none | none |
-
-
-Labels are shared across your entire organization. Any queue can use any label, and changes to a label's settings apply everywhere the label is used. Deleting a label does not remove existing scores that were created with it.
-
-
-
-Start with a few simple labels (e.g. a 5-star quality rating and a categorical sentiment label) before creating complex ones. You can always duplicate and customize later.
-
-
----
-
-## Next Steps
-
-
-
- Set up annotation queues that use your labels.
-
-
- Learn how to use labels in the annotation workspace.
-
-
- How label values are stored as scores and queried via the API.
-
-
diff --git a/src/pages/docs/annotations/features/queues.mdx b/src/pages/docs/annotations/features/queues.mdx
deleted file mode 100644
index 3bd1c067..00000000
--- a/src/pages/docs/annotations/features/queues.mdx
+++ /dev/null
@@ -1,168 +0,0 @@
----
-title: "Annotation Queues: Setup & Management"
-description: "Create annotation queues with round-robin or load-balanced assignment, multi-annotator support, reservation timeouts, and review workflows."
----
-
-## About
-
-An annotation queue is a managed campaign that groups items to annotate, assigns them to annotators, tracks progress, and enforces quality controls. Queues sit between labels (what to measure) and scores (the resulting data), providing the operational layer that turns annotation from an ad-hoc activity into a structured workflow.
-
----
-
-## Creating a Queue
-
-
-
- Go to **Annotations** in the left sidebar, then open the **Queues** tab.
-
- 
-
-
-
- Click the **Create Queue** button to open the creation form.
-
-
-
- Fill in the **Name** field (required) and an optional **Description** to help your team understand the queue's purpose.
-
-
-
- Select which annotation labels annotators will use when reviewing items in this queue. You can add as many labels as needed.
-
- 
-
-
-
- Select workspace members who will annotate items. Only selected members can access and annotate items in this queue.
-
-
-
- | Setting | Options | Default |
- |---------|---------|---------|
- | Annotations Required | 1-10 annotators per item | 1 |
- | Assignment Strategy | Manual, Round Robin, Load Balanced | Manual |
- | Reservation Timeout | 15 min, 30 min, 1 hour, 4 hours | 30 min |
- | Require Review | On / Off | Off |
-
-
-
- Write markdown-formatted guidelines for annotators. These appear in a collapsible panel in the annotation workspace. Use guidelines to define criteria, provide examples of correct/incorrect annotations, specify when to skip, and link to reference material.
-
-
-
- Click **Save**. The queue is created in **Draft** status. Add items and review settings before activating it.
-
-
-
----
-
-## Assignment Strategies
-
-| Strategy | Behavior | Best For |
-|----------|----------|----------|
-| **Manual** | Annotators browse and pick items themselves from the queue list. | Small queues or exploratory annotation where annotators need context to choose. |
-| **Round Robin** | Items are distributed cyclically across annotators in rotation. | Even distribution when annotators work at similar speeds. |
-| **Load Balanced** | Items are distributed based on each annotator's current workload. | Teams with varying availability or part-time annotators. |
-
----
-
-## Multi-Annotator Support
-
-For tasks that benefit from agreement between multiple reviewers, set the **Annotations Required** field (1-10).
-
-- Each item must receive the configured number of complete annotations before it transitions to **Completed**.
-- Different annotators independently annotate the same item. They do not see each other's responses.
-- The queue analytics tab shows inter-annotator agreement metrics once multiple annotators have scored the same items.
-
-
-An item is considered fully annotated by a single annotator only when all labels attached to the queue have been scored. Partial submissions are saved but do not count toward the required annotation count.
-
-
----
-
-## Reservation System
-
-When an annotator opens an item, the system reserves it for a configurable timeout period. This prevents two annotators from working on the same item simultaneously.
-
-- **Default timeout**: 30 minutes
-- **Configurable range**: 15 minutes to 4 hours
-- **Expiry behavior**: If the annotator does not submit or skip within the timeout, the reservation expires and the item returns to **Pending** for another annotator
-
----
-
-## Review Workflow
-
-Enable **Requires Review** on a queue to add a review step after annotation:
-
-1. Annotators complete their work as usual. When all required annotations are submitted, the item moves to **Pending Review** instead of **Completed**.
-2. A designated reviewer opens the item, sees all submitted annotations, and either **Approves** (moves to Completed) or **Rejects** (sends back to Pending for re-annotation).
-
-This is useful for high-stakes labeling tasks where a senior reviewer must validate annotations before they become final.
-
----
-
-## Queue Lifecycle
-
-| Status | Description | Can transition to |
-|--------|-------------|-------------------|
-| Draft | Queue is being set up, not yet accepting annotations | Active |
-| Active | Annotators can annotate items | Paused, Completed |
-| Paused | Temporarily stopped, no new annotations allowed | Active, Completed |
-| Completed | All items done or manually completed | Active (re-open) |
-
-### Activating a queue
-
-A newly created queue starts in **Draft**. To begin accepting annotations, use the menu and select **Activate**, or open the queue detail page and change the status.
-
-### Auto-completion
-
-Items auto-complete when:
-1. All labels attached to the queue have been scored for the item
-2. The required number of annotators have each fully annotated the item
-3. If **Requires Review** is enabled, the reviewer has approved the item
-
-
-When a completed queue receives new items, it automatically transitions back to **Active** so annotators can continue.
-
-
----
-
-## Item Statuses
-
-| Status | Meaning |
-|--------|---------|
-| **Pending** | Waiting for an annotator to pick it up |
-| **In Progress** | An annotator has opened the item and is actively annotating |
-| **Completed** | All required annotations have been submitted |
-| **Skipped** | An annotator chose to skip this item. It remains available for others. |
-| **Pending Review** | Annotations are done but awaiting reviewer approval (when review workflow is enabled) |
-
----
-
-## Managing Queues
-
-| Action | How |
-|---|---|
-| Edit | Open the queue detail page, use the **Settings** tab to modify name, labels, annotators, or workflow settings |
-| Duplicate | Use the menu and select **Duplicate**. Creates a copy in Draft status. |
-| Archive | Use the menu and select **Archive**. Soft-deletes the queue. |
-| Search and filter | Use the search bar to filter by name and the status dropdown to filter by queue status |
-
----
-
-## Next Steps
-
-
-
- Populate your queue with traces, sessions, dataset rows, and more.
-
-
- Walk through the annotation workspace and keyboard shortcuts.
-
-
- Track progress, annotator performance, and inter-annotator agreement.
-
-
- How annotation values are stored and queried.
-
-
diff --git a/src/pages/docs/annotations/guides/annotate-items.mdx b/src/pages/docs/annotations/guides/annotate-items.mdx
new file mode 100644
index 00000000..0481dcac
--- /dev/null
+++ b/src/pages/docs/annotations/guides/annotate-items.mdx
@@ -0,0 +1,81 @@
+---
+title: "Annotate items"
+description: "The step-by-step workflow annotators follow inside an annotation queue"
+---
+
+This is the annotator's view: opening a [queue](/docs/annotations/concepts/queues-and-items), working through its items one at a time, and submitting an answer for every [label](/docs/annotations/concepts/labels) it carries. Everything below plays out inside the annotation workspace itself.
+
+## Open the workspace
+
+Get in the same way [Explore a queue](/docs/annotations/guides/explore-queue) describes: click **Start Annotating** (or **Resume Skipped**) from the queue, or open an item directly from the Items tab. Either way lands you in the workspace on a specific item.
+
+### If you land on a message instead
+
+Sometimes the workspace hands you back a message instead of an item:
+
+- **Item Reserved**, when someone else already has that item open. Click **Skip to Next Item** to move on
+- **Queue Not Active**, if the queue isn't active. A manager has to reactivate it, see Explore a queue for where
+- **Assigned to {`{name}`}**, if the item belongs to someone else, a queue manager controls that assignment. Click **Skip to Next Item** to move on
+- **All Done!**, once there's nothing left assigned to you
+
+## Read the source
+
+The workspace splits into two resizable panes: the source on the left, the labels on the right. Drag the divider between them if you want more room for either side. The left pane renders whatever the item points to, a trace, a span, a session, a dataset row, a prototype run, or a voice call, so you can judge it before answering anything on the right.
+
+## Answer the labels
+
+If the queue's creator wrote instructions, they sit in a collapsible section above the labels, open by default so you see them on your first item. Once you know the guidance, collapse it to get it out of the way.
+
+The right pane lists every label the queue carries under a **Labels** heading. What each label's control looks like depends on its type. Labels covers what each type is for and [Label types & values](/docs/annotations/reference/label-types-and-values) has the exact constraints.
+
+You can't submit until every label has an answer. Leave one blank and the submit button stays disabled; nothing more happens until you try to submit. Press Cmd/Ctrl+Enter and a reminder names exactly which labels are still missing.
+
+Numeric and text answers are also checked against the label's configured bounds. Even if the control lets a value slip past, the check still runs again on the server when you submit. Anything out of range gets rejected; bring it back within the label's min, max, step, or length and submit again.
+
+## Add a note
+
+Below the labels, an optional **Notes** field lets you leave free-text context on the item ("Add notes for this item..."). It's your own commentary alongside the labels, not a label itself.
+
+## Submit
+
+The submit button's text tells you what submitting will actually do:
+
+- **Submit & Next** is the default: save your answers and move to the next item
+- **Submit for Review** shows instead when the queue requires reviewer approval, your answers go to a reviewer before the item counts as done
+- **Update & Next** shows when you're revising an answer you already submitted
+
+If a reviewer sends an item back with feedback, it shows up in the header's **Comments** button and as a **Reviewer feedback** alert at the top of the right pane. See [Review submissions](/docs/annotations/guides/review-submissions).
+
+## Skip an item
+
+Click **Skip** in the header, or press **S**. A skipped item isn't marked complete, it just steps out of your way so you can come back to it later. Skip is disabled once an item is completed or already pending review. On a completed item the button's tooltip flips to explain why; on a pending-review item it's just greyed out.
+
+## Move between items
+
+Move through the queue with these controls:
+
+- **Previous** and **Next** in the footer, alongside a position indicator (`n / total`)
+- **Back to Queue** in the header, which takes you out of the workspace and back to the queue's detail view
+- **Show completed**, which toggles whether items you've already finished are included as you move through the list
+
+## Keyboard shortcuts
+
+| Key | Action |
+|---|---|
+| Tab / Shift+Tab | Move between labels |
+| 1-9 | Quick-select a categorical option or star rating |
+| Cmd/Ctrl+Enter | Submit and move to the next item |
+| S | Skip the current item |
+| ← / → | Previous / next item |
+| ? | Toggle this shortcuts overlay |
+
+## Dive deeper
+
+
+
+ What happens to your answers once a reviewer looks at them
+
+
+ Score a single item on the spot, no queue required
+
+
diff --git a/src/pages/docs/annotations/guides/annotate-without-a-queue.mdx b/src/pages/docs/annotations/guides/annotate-without-a-queue.mdx
new file mode 100644
index 00000000..d3a1063c
--- /dev/null
+++ b/src/pages/docs/annotations/guides/annotate-without-a-queue.mdx
@@ -0,0 +1,40 @@
+---
+title: "Annotate without a queue"
+description: "Score a source on the spot, right from its own view, with no queue involved"
+---
+
+Not every [score](/docs/annotations/concepts/scores) needs a [queue](/docs/annotations/concepts/queues-and-items) behind it. If you already have a [source](/docs/annotations/concepts/understanding-annotation) open somewhere in Future AGI, a trace, span, dataset row, or prompt run, and something's worth flagging, score it right there instead of setting one up.
+
+## Open the source and find the Annotations tab
+
+Open the source's detail view and go to the **Annotations** tab. It lists whatever's already been scored on that source.
+
+## Switch to edit mode and pick your labels
+
+Click **Annotate** to switch into edit mode, then pick the labels that apply and fill in a value for each. The labels on offer are the ones set up for that workspace. See [Labels](/docs/annotations/concepts/labels) for how labels get organized.
+
+## Add a note, if a label supports it
+
+Not every label carries a note field, only the ones set up for it. Where one is available, it sits right under the label as its own box, already visible, no click needed. Type into it to add context that doesn't fit into the value itself.
+
+## Save to record the score
+
+Click **Save**. This writes a score exactly like one submitted through a queue, minus the queue-item link. It shows up wherever the source appears and in exports, right next to any queue-based scores on the same source.
+
+## When to reach for a queue instead
+
+A score saved on the spot is right for a one-off: you noticed something while looking at a source and want it on record. It stops being enough once more than one person needs to score the same batch of sources, or you need to see how far through that batch you've gotten. That's what a queue is for: it organizes who scores what, and whether a reviewer has to approve before it counts as done. [Create a queue](/docs/annotations/guides/create-queue) walks through setting one up.
+
+## Dive deeper
+
+
+
+ The identity a score carries, in or out of a queue
+
+
+ Set one up when scoring turns into a coordinated campaign
+
+
+ Write the same scores from the API instead of the UI
+
+
diff --git a/src/pages/docs/annotations/guides/create-label.mdx b/src/pages/docs/annotations/guides/create-label.mdx
new file mode 100644
index 00000000..842e3049
--- /dev/null
+++ b/src/pages/docs/annotations/guides/create-label.mdx
@@ -0,0 +1,70 @@
+---
+title: "Create a label"
+description: "Pick a type, configure its settings, and save a label ready to attach to a queue"
+---
+
+This walks through creating a label: a categorical label called `Response quality` with the options `Good`, `Needs work`, and `Wrong`. A [label](/docs/annotations/concepts/labels) needs a name, a type, and that type's settings.
+
+## Open the Labels tab
+
+Labels aren't tied to a project. Go to **Annotations** in the sidebar and switch to the **Labels** tab. Click **Create Label** to open the form.
+
+If this is your first label, the tab shows an empty state instead of a table, "No labels created yet." The same **Create Label** button sits there too.
+
+## Name it and pick a type
+
+Fill in **Name** and an optional **Description**. Those greyed-out examples in the Name field, `Relevance`, `Tone`, `Accuracy`, are placeholder text, not a fixed list to pick from, so `Response quality` fits just as well.
+
+Then pick a **Type**:
+
+- **Categorical**: a predefined set of options to choose from
+- **Numeric**: a score within a range
+- **Text**: free-text feedback
+- **Star Rating**: a star-based rating
+- **Thumbs Up/Down**: binary feedback
+
+Not sure which fits? Labels covers what each type is for and when to reach for it. Pick **Categorical** for `Response quality`.
+
+
+*The Create Label form, with Categorical selected as the type*
+
+Building one of the other four types instead? [Label types & values](/docs/annotations/reference/label-types-and-values) lists every setting, its default, and its validation rule by type.
+
+
+The type is locked once you save. Editing a label later lets you change its name, description, and settings, but not its type, so get this one right before you save.
+
+
+## Configure the settings and save
+
+Type-specific settings appear once you've picked a type. For Categorical, that's a list of options: add `Good`, `Needs work`, and `Wrong`.
+
+
+A categorical label needs at least two distinct, non-empty options. Duplicates aren't allowed either, `Good` and `good` count as the same option. The form doesn't stop you from entering duplicates: clicking **Create** fails with an error toast, so fix the options and click **Create** again.
+
+
+Check **Allow notes** if you want annotators to attach free-text commentary alongside their `Response quality` value. Leave it unchecked for a label that should stay a single, quick choice.
+
+Click **Create**. `Response quality` is now available to attach to a [queue](/docs/annotations/concepts/queues-and-items) or score an item [inline](/docs/annotations/guides/annotate-without-a-queue) (on the spot, with no queue involved).
+
+## Managing labels
+
+The same tab handles the rest, once you have labels to manage:
+
+- **Edit** a label to change its name, description, or settings
+- **Duplicate** a label to open the form pre-filled with its settings under a new name, so you can adjust and save it as a separate label
+- **Archive** a label to take it out of active use, and **Restore** it from the **Archived** view when you need it back
+- **Search** narrows the list by name, and the **type** filter shows only labels of one type
+
+## Dive deeper
+
+
+
+ Score an item inline, on the spot, with no queue involved
+
+
+ Every settings key, default, and validation rule by type
+
+
+ Attach your new label to a queue and start collecting scores
+
+
diff --git a/src/pages/docs/annotations/guides/create-queue.mdx b/src/pages/docs/annotations/guides/create-queue.mdx
new file mode 100644
index 00000000..753ff222
--- /dev/null
+++ b/src/pages/docs/annotations/guides/create-queue.mdx
@@ -0,0 +1,77 @@
+---
+title: "Create a queue"
+description: "Fill out the create-queue drawer field by field, then activate the queue so annotators can start"
+---
+
+A [queue](/docs/annotations/concepts/queues-and-items) attaches one or more [labels](/docs/annotations/concepts/labels) to a set of items, adds the people who'll answer them, and sets how many independent submissions each item needs. This walks through the create-queue drawer in the order it presents its fields, building one running example: `Support quality review`, carrying the `Response quality` label with two submissions per item.
+
+
+- You need at least one label before you open this drawer. Queues won't save without one. Build `Response quality` first if you haven't; see [Create a label](/docs/annotations/guides/create-label)
+- This example needs one other workspace member picked as an annotator alongside you, since submissions per item can't exceed the number of annotators picked there
+- Invite them from [User Management](/docs/admin-settings/user-management) if they're not in your workspace yet
+- Working solo? Leave your own row as it is and set submissions per item to 1
+
+
+## Open the drawer
+
+Go to **Annotations** in the sidebar, switch to the **Queues** tab, and click **Create Queue** to open the drawer. A queue can belong to a project, a dataset, or an agent definition, or to none of them at all, an org-level queue. This drawer doesn't ask you to pick one, so `Support quality review` comes out org-level.
+
+## Name it and describe it
+
+**Queue Name** is required and free text: type `Support quality review`. The name has to be unique, case-insensitively, among the queues you haven't archived in that same scope, so a name already used there is rejected.
+
+**Description** is optional, a line or two on the queue's purpose.
+
+## Attach labels
+
+Pick the labels annotators will answer for every item in this queue. At least one is required. The drawer won't let you save without it. Attach `Response quality`.
+
+## Add annotators
+
+Add the workspace members who'll work this queue as annotators. The picker opens with your own row already selected as Annotator, Reviewer, and Manager, tagged (creator), so you already count toward submissions per item. Untick Annotator on your row if you don't want to answer items yourself. For `Support quality review`, add one more member: you and them make two.
+
+**Auto-assign items to all annotators** is a checkbox alongside the picker. Turn it on and every annotator is assigned to every item, so anyone can open anything. Leave it off and assignment is manual instead, which means someone has to hand items out (covered in [Add items](/docs/annotations/guides/explore-queue/add-items)).
+
+## Set submissions per item
+
+**Submissions per item** is how many different annotators have to answer each item before it's done. It can't exceed the number of annotators you've added. Set it to `2` for `Support quality review`, so every item needs two independent takes on `Response quality` before it's complete.
+
+## Write instructions
+
+**Instructions** is a free-text field, markdown supported, that shows up in the annotation workspace itself. Optional, but worth using for queue-wide guidance that applies across every item and label in the queue, rather than notes on a single label. For `Support quality review`, something like:
+
+```markdown
+Rate the agent's final response, not the whole conversation.
+
+- **Good**: fully answers the question and matches our tone guidelines
+- **Needs work**: correct but incomplete, robotic, or missing context
+- **Wrong**: factually incorrect or answers a different question than the one asked
+```
+
+## Open Advanced settings
+
+Advanced settings is collapsed by default and holds three fields:
+
+- **Assignment strategy**: Manual is the only one you can pick today. Round Robin and Load Balanced are visible with a **Coming soon** chip, disabled
+- **Reservation timeout**: how long an item stays reserved for the annotator who opened it, one of 15 minutes, 30 minutes, 1 hour (the default), or 4 hours. When it expires, the item is released back to the queue for another annotator to pick up
+- **Require reviewer approval**: a gated feature that needs the review workflow entitlement. Turn it on and a fully annotated item lands in review instead of finishing outright, which [Review submissions](/docs/annotations/guides/review-submissions) covers; without the entitlement, turning it on fails with an upgrade prompt
+
+## Save, then activate it
+
+Click **Create annotation queue**. It saves in Draft, and Draft can only move to Active, nowhere else. Annotating doesn't start until you make that move: open the queue's row menu and click **Activate**.
+
+Activating doesn't add any items either: the queue starts empty, add them next. From Active, a queue can move to Paused and on to Completed; see [Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits) for the exact transitions.
+
+## Dive deeper
+
+
+
+ Orient yourself in the queue you just created
+
+
+ Hand-pick items or add them by filter from the Items tab
+
+
+ Every field, status, role, and cap the queue carries
+
+
diff --git a/src/pages/docs/annotations/guides/explore-queue/add-items.mdx b/src/pages/docs/annotations/guides/explore-queue/add-items.mdx
new file mode 100644
index 00000000..54b993a3
--- /dev/null
+++ b/src/pages/docs/annotations/guides/explore-queue/add-items.mdx
@@ -0,0 +1,70 @@
+---
+title: "Add items"
+description: "Pull sources into a queue and hand them to the right people"
+---
+
+A [queue](/docs/annotations/concepts/queues-and-items)'s **Items** tab is empty until you put something in it. This guide covers getting items into `Support quality review`, the running example from [Create a queue](/docs/annotations/guides/create-queue), then filtering and assigning them once they're there. Adding and assigning are both manager work: if you're not a manager on the queue, you won't see these controls at all.
+
+## Add items
+
+Open `Support quality review` and switch to the **Items** tab. A queue with nothing in it shows **No items in this queue** with its own **Add Items** button. Once the queue has items, that button moves into the toolbar above the item table, next to the item filters covered below. Either way, click **Add Items** to open the picker: two ways to fill the queue, hand-pick specific items or set a filter and add everything it matches. Hand-pick when you already know the exact items you want; use filter mode when you want everything matching a rule, however many that turns out to be.
+
+You can also push items from the source instead of pulling them from the queue. Select rows in a [traces](/docs/observe) table, then **Actions > Add to annotation queue** opens a popover listing your queues: search for one, pick it, or create a new queue on the spot.
+
+### Hand-pick items
+
+1. Choose where the items come from: **From Datasets**, **From Traces**, **From Spans**, **From Sessions**, or **From Simulation**
+2. Use the checkbox column to select the specific rows you want
+3. Click **Add to queue**
+
+### Add by filter
+
+1. Choose a source the same way as hand-picking
+2. Set the filters that describe what you want, using the filter controls above that source's table
+3. Instead of checking rows one by one, tick the header checkbox, then click **Select all N matching your filter** in the banner that appears above the table
+4. Click **Add to queue**. For traces, spans, sessions, and simulation, the queue resolves everything the filter matches on the server at that moment, not just what's loaded on screen
+
+
+For a hand-picked selection, the picker splits a large add into multiple requests for you. Filter mode selections are capped at 10,000 items: past that, the whole add is rejected and you're asked to narrow the filter first. Call the endpoint yourself instead of using the picker, and an enumerated list over 1,000 items is rejected outright with an HTTP 413. Full caps and gated behavior live on [Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits).
+
+
+Two things worth knowing about what happens once items are added:
+
+- **Duplicates are skipped, not errored.** If a source is already in the queue, or shows up twice in what you're adding, it's dropped, and the confirmation names the count, for example "12 items added · 3 already in queue"
+- **Assignment can happen automatically.** If the queue has auto-assign turned on, incoming items get an assignee the moment they land. Otherwise they arrive unassigned, and someone has to assign them by hand
+
+
+The preview shown in the Items table is a snapshot captured the moment the item was added, not a live read of the source. Opening the item to annotate it always shows the current source.
+
+
+## Filter and find items
+
+Above the item table sit the item filters:
+
+- Item status
+- Source
+- Review status, shown only when the queue has review turned on
+
+There's also a **My Items** toggle that narrows the table down to whatever's assigned to you.
+
+## Assign and remove items
+
+Select one or more rows and buttons join that same toolbar, scoped to your selection: **Assign Selected** and **Remove Selected**.
+
+**Assign Selected** opens the Assign Selected Items dialog, and like adding items, it's manager-only. Whoever you check becomes the full set of assignees on the selected items, replacing whatever was there before; check nobody and it clears the assignment instead. You can only check people who are already members of the queue: you can't assign an item to someone who hasn't been added yet.
+
+**Remove Selected** takes the selected items out of the queue entirely, along with any annotations already submitted on them. There's no undo: getting an item back means adding it again as a new item, starting from zero submissions.
+
+## Dive deeper
+
+
+
+ Work through what you just added
+
+
+ See how the queue is filling up
+
+
+ Every cap and gated feature in one place
+
+
diff --git a/src/pages/docs/annotations/guides/explore-queue/automate-item-intake.mdx b/src/pages/docs/annotations/guides/explore-queue/automate-item-intake.mdx
new file mode 100644
index 00000000..c6542b82
--- /dev/null
+++ b/src/pages/docs/annotations/guides/explore-queue/automate-item-intake.mdx
@@ -0,0 +1,60 @@
+---
+title: "Automate item intake"
+description: "Build a rule that automatically feeds a queue with matching items"
+---
+
+A rule checks new candidates against conditions you set and adds the matches to a [queue](/docs/annotations/concepts/queues-and-items) on its own, so nobody has to go looking for fresh items to hand out. This walks through building one for `Support quality review`, the running example from [Create a queue](/docs/annotations/guides/create-queue), running it once to see what it catches, and what actually happens once you trigger it. Creating, editing, deleting, and running a rule are all manager work: if you're not a manager on the queue, following these steps just gets you an error. If you'd rather add items yourself, see [Add items](/docs/annotations/guides/explore-queue/add-items).
+
+## Create a rule
+
+Only queue managers can create or run rules. If you're not one, step 1 below returns "Only queue managers can manage automation rules." instead of opening the form.
+
+From `Support quality review`'s **Rules** tab, click **Add Rule** in the top right to start a new rule.
+
+1. Give it a name
+2. Choose a **Source type** to say what kind of candidate the rule looks at, one of:
+ - Dataset Row
+ - Trace
+ - Span
+ - Session
+ - Simulation
+3. Set the **Trigger**, one of:
+ - **Manually**: the rule never fires on its own, someone has to run it every time
+ - **Every hour**, **Daily**, **Weekly**, or **Monthly**: puts it on a recurring schedule instead
+4. Pick the specific target the rule reads from, one of:
+ - Dataset
+ - Project
+ - Agent Definition
+
+ If the queue is already scoped to a dataset or project, that field is locked and reads "Locked by this queue"
+5. Add the conditions that decide which candidates match under **Conditions**. The available fields and operators depend on the Source type you picked
+6. Click **Create Rule**. It stays disabled until the rule has a name and a source
+
+Once a rule exists, click its row to open **Edit Automation Rule** and change its name, source type, conditions, or trigger. Delete is the **x** at the end of the row; it asks for confirmation first.
+
+## Run it once before you trust it
+
+Whatever trigger you picked, click **Run Now** on the rule's row in the Rules tab to see what it catches from its source right now, before you let it run unattended. The first run always scans the whole backlog against the conditions, whether it fires by schedule or because you clicked Run Now. After that first run, a scheduled rule only rescans what's new since it last ran, so Run Now on a rule that's already fired once is a smaller check, not a full rescan; a Manually-triggered rule has no schedule to fall back on, so every run stays a full rescan. Run Now stays disabled until the rule is enabled with the **Enabled** switch in the rule's row; its tooltip reads "Enable this rule before running it". Click **Run Now** again while a run is still going and it's refused with "A run is already in progress for this rule".
+
+## When a rule runs
+
+This is the part that trips people up:
+
+- A **scheduled** rule (Every hour, Daily, Weekly, Monthly) doesn't fire at the exact minute its trigger implies, so treat the trigger as "within about an hour of," not "on the dot"
+- **Running a rule yourself** either reports what it added in the toast right away, or shows "We're preparing your data" and finishes in the background
+- When a run finishes in the background, the person who ran it, the rule's creator, and every manager on the queue get an email. Scheduled runs don't send it
+
+## New items don't arrive silently
+
+Even without that email, annotators find out. Everyone gets a new-item email at most once an hour, plus a daily summary at their own local digest hour, unless they've snoozed notifications. Whatever a rule adds to `Support quality review` reaches annotators on both cadences, so a rule firing while you're not watching still gets to the people who need to work the items.
+
+## Dive deeper
+
+
+
+ Work through what the rule and your team add
+
+
+ See where the queue's items stand once they're flowing in
+
+
diff --git a/src/pages/docs/annotations/guides/explore-queue/index.mdx b/src/pages/docs/annotations/guides/explore-queue/index.mdx
new file mode 100644
index 00000000..f97f0f55
--- /dev/null
+++ b/src/pages/docs/annotations/guides/explore-queue/index.mdx
@@ -0,0 +1,63 @@
+---
+title: "Explore a queue"
+description: "Find your way around a queue's detail view: its header, tabs, and toolbar actions"
+---
+
+Once a [queue](/docs/annotations/concepts/queues-and-items) has items and annotators in it, this detail view is where the work actually happens. This guide walks that view using `Support quality review`, the queue built earlier, as the example; any queue of your own works the same way.
+
+These guides all start from a queue you've already created. If you don't have one yet, [Create a queue](/docs/annotations/guides/create-queue) makes the first one.
+
+## Open the queue
+
+Go to **Annotations → Queues** and click the `Support quality review` row to open it.
+
+## Read the header
+
+The header carries the queue's name, a status badge next to it, progress underneath, and a row of actions on the right, covered below.
+
+The badge reads **Draft**, **Active**, **Paused**, or **Completed**. It's the one place to check whether the queue is unlocked for annotating before you try to open an item.
+
+If you have items assigned to you, you get two progress bars: your own progress first, then the queue's overall progress with a breakdown of how many items are pending, in progress, in review, or skipped.
+
+## The five tabs
+
+Below the header, the view splits into five tabs:
+
+- **Items**, the list of items in the queue and where you open one to annotate it
+- **Settings**, the queue's configuration, managers only
+- **[Analytics](/docs/annotations/guides/explore-queue/progress-and-agreement)**, performance across the queue
+- **Agreement**, how consistently annotators score the same items
+- **[Rules](/docs/annotations/guides/explore-queue/automate-item-intake)**, automation rules that feed items into the queue, managers only
+
+**Settings** and **Rules** only show up if you're a manager on the queue. Everyone else sees Items, Analytics, and Agreement.
+
+## The toolbar
+
+Four actions live in the header toolbar. Two of them switch labels depending on the queue's state, and which ones you see at all depends on your role:
+
+- **Activate** shows up for managers only, and only while the queue isn't already active
+- **Export** opens a menu with **Download** and **Export to Dataset**. It's grayed out until the queue has items in it
+- **Review Items** is for reviewers and managers, and shows up once the queue has items and is active or completed. If the queue doesn't [require review](/docs/annotations/guides/create-queue#open-advanced-settings), this button reads **View Submissions** instead
+- **Start Annotating** opens the annotation workspace for anyone who can annotate, picking the next available item for you. Once the queue is complete and has skipped items left, this button reads **Resume Skipped** instead. Opening an item straight from the **Items** tab lands you in that same workspace, just on the item you clicked instead of the next one in line
+
+## The two rules for annotating an item
+
+- **You can only annotate while the queue is active, except to resume skipped items once it's completed.** Try to open an item outside that state and you're told to manage the status from the Settings tab first. If you're a manager, the toolbar's **Activate** button does the same thing in one click whenever the queue isn't already active; use Settings for any other status change, or if you don't see the button
+- **You can only open an item assigned to you, unless auto-assign is on, or you're a manager or reviewer on the queue.** Auto-assign is a checkbox in the queue's Settings tab, the same one you set when you built the queue. For plain annotators, an item assigned to someone else stays closed even if you can see it in the list
+
+## Dive deeper
+
+
+
+ Open the workspace and work through the labels
+
+
+ Put more items in front of your annotators
+
+
+ Read the Analytics and Agreement tabs
+
+
+ Set up rules so items feed into the queue on their own
+
+
diff --git a/src/pages/docs/annotations/guides/explore-queue/progress-and-agreement.mdx b/src/pages/docs/annotations/guides/explore-queue/progress-and-agreement.mdx
new file mode 100644
index 00000000..3be9ee41
--- /dev/null
+++ b/src/pages/docs/annotations/guides/explore-queue/progress-and-agreement.mdx
@@ -0,0 +1,82 @@
+---
+title: "Track progress & agreement"
+description: "Track how a queue's work is coming along, including annotator agreement"
+---
+
+The **Analytics** and **Agreement** tabs sit after **Items** on a queue's detail view (managers also see **Settings** there). Analytics tells you how the work is going: how much is done, how fast, and who's doing it. Agreement tells you something Analytics can't: whether two people looking at the same item score it the same way. This walks both tabs using `Support quality review`, the queue from [Create a queue](/docs/annotations/guides/create-queue), which carries the `Response quality` label and two submissions required per item.
+
+## Read the Analytics tab
+
+Open `Support quality review` and switch to the **Analytics** tab.
+
+### Headline numbers
+
+Four cards summarize the queue at a glance:
+
+- **Total Items**: a raw count of everything in the queue
+- **Completed**: a raw count of items that are done
+- **Completion Rate**: Completed turned into a percentage of Total Items
+- **Avg / Day**: completions averaged over the last 30 days
+
+### Status breakdown
+
+Below the headline cards, a bar breaks total items into six buckets, each with its own count:
+
+- **Completed**
+- **In Review**
+- **Needs Changes**
+- **Resubmitted**
+- **Pending Annotation**
+- **Skipped**
+
+Needs Changes and Resubmitted come out of the [review workflow](/docs/annotations/guides/review-submissions): an item only moves through them when the queue requires reviewer approval. In Review holds both: items part-way to the queue's required submissions, and items that have already reached those submissions on a review-enabled queue but are still waiting on a reviewer's verdict. On `Support quality review`, which requires two submissions per item, the first annotator's submission alone puts the item in In Review, no reviewer needed.
+
+### Throughput over time
+
+**Daily Throughput (Last 30 Days)** charts completions per day over that same window. Use it to spot a slowdown, or to confirm that adding annotators actually moved the queue faster.
+
+### Label distribution
+
+**Label Distribution** shows one card per label, breaking down every value annotators have submitted for it. For `Response quality`, that's a bar for each option, `Good`, `Needs work`, `Wrong`, with a count of how many times annotators picked it. A numeric or star label shows the same idea by rating instead of option, and a thumbs label shows up versus down ([Label types & values](/docs/annotations/reference/label-types-and-values) covers every type).
+
+### Annotator performance
+
+**Annotator Performance** lists everyone with activity in the queue. Completed counts items that have met the queue's required submissions across every required label, so it credits every annotator whose submission contributed to that item, not just the one who finished it.
+
+## Read the Agreement tab
+
+Switch to the **Agreement** tab.
+
+
+Agreement only has something to compare once two different annotators have actually scored the same item, not just been assigned to it. That means the queue's submissions-per-item setting has to be above its default of 1 ([Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits) covers where that lives), with submissions from two or more people on the same item. `Support quality review` is already set to 2, so its Agreement tab fills in as soon as a second annotator submits on the same item.
+
+Agreement is also a gated feature that needs the Agreement Metrics entitlement: without it, the tab comes up empty instead of loading.
+
+
+### Overall agreement
+
+At the top, **Overall Agreement** shows one percentage: the share of item/label pairs where every annotator who scored it landed on the same value. Until the precondition above is met, the number reads N/A, with "Need at least 2 annotators per item to calculate agreement" underneath it.
+
+### Per-label agreement
+
+**Per-Label Agreement** breaks that number down one row per label. Agreement here is the same raw percentage, scoped to that one label; Disagreements counts how many items its annotators didn't match on.
+
+Cohen's Kappa is the only agreement statistic the tab reports, whether two annotators scored an item or five. It isn't swapped for a different multi-rater statistic once a third annotator joins. Kappa corrects the raw Agreement percentage for how often annotators would land on the same value purely by chance, so it usually reads lower, and more honestly, than Agreement alone, especially on a label with few options. As a rough guide: below 0.20 is poor agreement, 0.21–0.40 fair, 0.41–0.60 moderate, 0.61–0.80 substantial, and above 0.80 almost perfect. Kappa only computes for categorical, numeric, star, and thumbs labels; a free-text label shows a dash instead.
+
+### Annotator pair agreement
+
+As soon as any single pair of annotators has overlapping work, **Annotator Pair Agreement** lists that pair with their agreement percentage and a Comparisons count: the number of item/label comparisons they share, not items, so one item with three labels counts as three. It's the fastest way to tell an annotator who disagrees with everyone else apart from a label that's just genuinely hard to agree on.
+
+## Dive deeper
+
+
+
+ Where submissions per item and every other queue field lives
+
+
+ Turn completed annotations into a dataset or a file
+
+
+ Keep the queue fed so Analytics has something to track
+
+
diff --git a/src/pages/docs/annotations/guides/export-annotations.mdx b/src/pages/docs/annotations/guides/export-annotations.mdx
new file mode 100644
index 00000000..ca7a7990
--- /dev/null
+++ b/src/pages/docs/annotations/guides/export-annotations.mdx
@@ -0,0 +1,64 @@
+---
+title: "Export annotations"
+description: "Download a queue's results as a file, or write them into a Future AGI dataset"
+---
+
+`Support quality review`, the [queue](/docs/annotations/concepts/queues-and-items) from [Create a queue](/docs/annotations/guides/create-queue), now has completed items worth keeping. This guide walks through both routes, starting with the quick download and ending with a write into a [dataset](/docs/dataset).
+
+## Download as JSON or CSV
+
+Open `Support quality review` and click **Export > Download** in the header. Every item in the queue, any status, downloads immediately as a JSON file: one entry per item, carrying its annotations, review status, and the source's own content resolved onto it.
+
+
+The endpoint behind Download also accepts a CSV format, which flattens item, review, and annotation fields to one row per label value. It drops `source`, `evals`, `source_id`, `item_notes`, and `annotation_metrics`. There's no format picker in the UI for it yet, so pull CSV directly through the API if you need rows instead of nested JSON; see [SDK & API](/docs/annotations/reference/sdk-api).
+
+
+
+Download tops out at 1,000 items. Push past that and it returns an error instead of a file, since there's no status filter on this button to narrow the set first. For a queue that big, use Export to Dataset instead, which carries no such cap, or filter by status through the API. Self-hosted deployments can raise the ceiling with the `ANNOTATION_EXPORT_SYNC_MAX` Django setting.
+
+
+### What's in an export
+
+| Field | What it holds |
+|---|---|
+| `item_id` | The queue item's ID |
+| `source_type` | trace, observation_span, trace_session, prototype_run, call_execution, or dataset_row |
+| `source_id` | ID of the annotated source |
+| `status` | pending, in_progress, completed, or skipped |
+| `order` | The item's position in the queue |
+| `review` | Review status and reviewer, filled in once the item's been reviewed |
+| `item_notes` | The latest note left on the item |
+| `annotations` | Every label value submitted, with the annotator and score source |
+| `annotation_metrics` | The item's annotations keyed by label name |
+| `evals` | Eval scores already attached to the item's source, if any |
+| `source` | The resolved content of the source itself |
+
+## Export to Dataset
+
+Click **Export > Export to Dataset** in the header to open the export drawer.
+
+1. Choose **Create new dataset** and name it, or **Add to existing dataset** and search for one
+2. Set **Items to export**: it defaults to Completed only, and can widen to All items, or switch to Pending only or In Progress only
+3. Review the column mapping: each source field, label, and review detail maps to a dataset column, and you can rename, add, or drop columns before running
+4. Click **Export**
+
+Unlike Download, Export to Dataset has no item cap, so a queue past 1,000 items still exports in full.
+
+## What you do with it
+
+- **Fine-tuning**: the annotated examples become training data for a model update
+- **Eval datasets**: completed items become a golden set you run other evals against
+
+## Dive deeper
+
+
+
+ What a dataset is and what you can run against one
+
+
+ Pull an export, including CSV, straight from the API
+
+
+ Every cap and gated feature in one place
+
+
diff --git a/src/pages/docs/annotations/guides/review-submissions.mdx b/src/pages/docs/annotations/guides/review-submissions.mdx
new file mode 100644
index 00000000..5f7dc1bd
--- /dev/null
+++ b/src/pages/docs/annotations/guides/review-submissions.mdx
@@ -0,0 +1,64 @@
+---
+title: "Review submissions"
+description: "Compare annotators' answers side by side, approve or send items back for changes, and follow the back-and-forth in comment threads"
+---
+
+This walks through the reviewer's side of a review-gated queue: reading what came in, approving or sending it back, and clearing a stack in bulk.
+
+`Support quality review`, the [queue](/docs/annotations/concepts/queues-and-items) from [Create a queue](/docs/annotations/guides/create-queue), requires reviewer approval. That changes what happens once an item collects its [`Response quality`](/docs/annotations/concepts/labels) submissions: instead of finishing on its own, a fully annotated item lands in pending review, and the button its annotators see reads **Submit for Review** rather than **Submit & Next** (covered in [Annotate items](/docs/annotations/guides/annotate-items)).
+
+## Get into review mode
+
+Reviewer approval is itself a gated feature: an org needs the entitlement before **Require reviewer approval** can be turned on for a queue at all ([Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits) has the details). Everything below assumes it's on.
+
+If you hold both the annotator and reviewer roles on the queue, the annotation workspace carries a **Workspace action** toggle: **Annotate my answers** next to **Review submissions**. The toggle only appears when you hold both roles, so a reviewer without the annotator role, or an annotator without the reviewer role, never sees it.
+
+If you only hold the reviewer role, open the queue from its list and use the **Review Items** button in the header instead, which opens the review workspace directly on the first item pending review.
+
+Whether that button reads **Review Items** or **View Submissions** depends on the queue's **Require reviewer approval** setting, not on the entitlement: a queue can have the review-workflow entitlement and still show **View Submissions** if that setting is off. With it off, the workspace is read-only, labelled **View submissions** instead of **Review submissions** in the toggle: you can read what was submitted, but there's nothing to approve or send back.
+
+## Compare answers side by side
+
+Opening an item that's pending review puts you in the comparison panel, which lays out every annotator's answer for the item side by side instead of one at a time, so you can weigh `Response quality` from both submissions before deciding. Each answer has its own feedback field for comments scoped to just that answer, and the panel carries the Approve and Request changes actions for the item as a whole.
+
+## Approve or request changes
+
+Two actions sit in the comparison panel: **Approve**, which moves the item to completed, and **Request changes**, which sends it back to the annotator.
+
+- **Approve** takes no note at all, and is disabled the moment you type any feedback into the panel
+- **Request changes** stays disabled until you've left a note in the **Whole-item feedback** box or targeted feedback on at least one answer
+- **Approve** is also unavailable while an earlier request for changes on the item is still open; it has to be addressed first before Approve is available again
+
+
+You can't review an item you annotated yourself. Approve and Request changes don't render at all on an item you submitted answers for, even if you also hold the reviewer role on the queue.
+
+
+Once the annotator resubmits, the item lands back in pending review the same way it did the first time, so it reappears in your Items tab list for another look.
+
+## Leave targeted feedback
+
+When the problem is one specific answer rather than the whole item, click **Feedback** on that annotator's row to open a feedback field scoped to that answer: **Clear** discards what you've typed, **Done** closes it. That's the targeted alternative to a whole-item note when only one annotator's answer needs fixing. Filling in this field, or the whole-item note, is what enables Request changes; clearing it back out is what re-enables Approve.
+
+## Approve in bulk
+
+When there's nothing to argue with, you don't have to open every item on its own. From the queue's **Items** tab, select the items you want to clear and click **Approve Selected**. It counts only the items in your selection that are pending review, and it only appears once at least one selected item is.
+
+## Comment threads
+
+Every review action (comment, approve, or request changes) drops into a thread scoped to the item or to one specific answer. A thread moves through open, addressed, resolved, and reopened as reviewers and annotators go back and forth.
+
+You can @mention teammates in a comment, up to 50 per comment.
+
+## Dive deeper
+
+
+
+ See how review activity shows up in the queue's analytics
+
+
+ Get the approved answers out as a file or a dataset
+
+
+ The reviewer role, the review-workflow entitlement, and the caps on comments
+
+
diff --git a/src/pages/docs/annotations/index.mdx b/src/pages/docs/annotations/index.mdx
index 77a17818..99358b04 100644
--- a/src/pages/docs/annotations/index.mdx
+++ b/src/pages/docs/annotations/index.mdx
@@ -1,88 +1,49 @@
---
-title: "Annotations: Human-in-the-Loop Feedback"
-description: "Capture human feedback on AI outputs using labels, queues, and scores across traces, spans, sessions, datasets, prototypes, and simulations."
+title: "Overview"
+description: "The three objects behind every human judgement on your AI's output"
---
-## About
+## What is Annotation?
-Annotations are human labels applied to AI outputs -- traces, spans, sessions, dataset rows, prototype runs, and simulation executions. They capture subjective judgments (sentiment, quality, helpfulness) and factual assessments (correctness, safety, relevance) that automated evals alone cannot provide.
+Annotation captures human judgement on AI output and stores every judgement as a score you can filter, export, and turn into a dataset. The judged thing can be a trace, a span, a session, a call execution, a prototype run, or a dataset row.
-Human-in-the-loop (HITL) feedback is essential for GenAI systems because:
+## Labels, queues, and scores
-- **Quality control** -- Catch hallucinations, off-topic responses, and policy violations before they reach users.
-- **Feedback loops** -- Route human judgments back into prompt tuning, guardrail configuration, and model selection.
-- **Fine-tuning data** -- Build high-quality labeled datasets from production traffic to improve your models.
-- **Safety and compliance** -- Document human review for regulated or high-stakes use cases.
+Three objects carry the whole model:
-## Architecture
+- A **[label](/docs/annotations/concepts/labels)** is the question you ask: a reusable definition of what you're judging, with a fixed answer type. `Response quality`, for instance, is categorical with three options: `Good`, `Needs work`, `Wrong`
+- A **[queue](/docs/annotations/concepts/queues-and-items)** organises who answers it and on what: a managed campaign that assigns items, the individual pieces of output being judged, to annotators and tracks their progress. Attach `Response quality` to a `Support quality review` queue and every annotator working it answers that same question
+- A **[score](/docs/annotations/concepts/scores)** is the answer itself, one record per judgement. An annotator answering `Response quality` on an item in `Support quality review` produces one score: `Good`
-Annotations are built on three primitives:
+You can produce a score two ways: work an item through a queue, or [annotate it inline](/docs/annotations/guides/annotate-without-a-queue), on the spot, with no queue involved.
-| Primitive | Purpose |
-|-----------|---------|
-| **Labels** | Reusable annotation templates (categorical, numeric, text, star rating, thumbs up/down) shared across your organization. |
-| **Queues** | Managed annotation campaigns that assign items to annotators, track progress, and enforce review workflows. |
-| **Scores** | The unified data record created each time an annotator (or automation) applies a label to a source. |
+## Start here
-Labels define *what* you measure. Queues organize *how* the work gets done. Scores store *every individual annotation*.
-
-## Supported source types
-
-Annotations can target any of the following entities:
-
-| Source Type | Description |
-|-------------|-------------|
-| `trace` | An LLM trace from Observe |
-| `observation_span` | A specific span within a trace |
-| `trace_session` | A conversation session (group of traces) |
-| `dataset_row` | A row in a dataset |
-| `call_execution` | A simulation call execution |
-| `prototype_run` | A prototype/experiment run |
-
-## How it works
-
-The typical annotation workflow follows three steps:
-
-1. **Define labels** -- Create the annotation templates your team will use (e.g. a "Sentiment" categorical label or a "Quality" star rating).
-2. **Set up a queue** -- Build an annotation campaign by choosing labels, adding annotators, and configuring assignment rules.
-3. **Annotate and review** -- Add items (traces, dataset rows, etc.) to the queue. Annotators score each item. Reviewers optionally approve results.
-
-Annotations can also be created **inline** -- directly from any trace, session, or dataset view -- without a queue, for ad-hoc feedback.
-
-## Key capabilities
-
-- **5 label types** -- Categorical, numeric, free-text, star rating, and thumbs up/down to cover any feedback need.
-- **Managed queues** -- Round-robin, load-balanced, or manual assignment strategies with reservation timeouts.
-- **Inline annotations** -- Annotate directly from trace detail, session grid, or dataset views without opening a queue.
-- **Multi-annotator support** -- Require 1-10 annotators per item for inter-annotator agreement.
-- **Review workflows** -- Route completed items through a reviewer before finalizing.
-- **Export to dataset** -- Turn annotated data into training or eval datasets.
-- **Python and JS SDK** -- Create labels, manage queues, and submit scores programmatically.
-
-## Common use cases
-
-| Use Case | Label Type | Example |
-|----------|------------|---------|
-| Sentiment classification | Categorical | Positive / Negative / Neutral |
-| Factual accuracy | Thumbs up/down | Correct vs. hallucinated |
-| Toxicity screening | Categorical | Safe / Borderline / Toxic |
-| Response relevance | Numeric (1-10) | How relevant was the answer? |
-| Grammar and style | Text | Free-form correction notes |
-| Prompt A vs. B comparison | Star rating | Rate each variant 1-5 stars |
+
+
+ Stand up an active queue and start collecting judgement
+
+
+ Work through a queue as an annotator
+
+
+ Score a single item on the spot, no queue involved
+
+
-## Get started
+## Concepts
-
- Create a label, set up a queue, and annotate your first item in 5 minutes.
+
+ The object model end to end: how labels, queues, items, and scores connect
-
- Understand the five label types and when to use each one.
+
+ The answer types and how to pick one
-
- Learn how queues organize work with assignment strategies and review workflows.
+
+ Roles, statuses, and how an item gets to complete
-
- Dive into the unified Score model that powers all annotation data.
+
+ What a score record carries, and when a new one is created instead of an edit
diff --git a/src/pages/docs/annotations/quickstart.mdx b/src/pages/docs/annotations/quickstart.mdx
deleted file mode 100644
index fcdfe958..00000000
--- a/src/pages/docs/annotations/quickstart.mdx
+++ /dev/null
@@ -1,94 +0,0 @@
----
-title: "Annotations Quickstart: Label & Queue"
-description: "Create an annotation label, set up a queue, add traces, and annotate your first item in 5 minutes with this Future AGI walkthrough."
----
-
-## What you will do
-
-In this walkthrough you will create an annotation label, set up a queue, add traces to it, and annotate your first item. The entire flow takes about 5 minutes.
-
-
-
- Navigate to **Annotations** in the left sidebar, then open the **Labels** tab. Click **Create Label**.
-
- 
-
- Fill in the form:
-
- | Field | Value |
- |-------|-------|
- | Name | `Sentiment` |
- | Type | Categorical |
- | Options | `Positive`, `Negative`, `Neutral` |
- | Allow Notes | Enabled |
-
- Click **Create** to save.
-
- 
-
-
-
- Switch to the **Queues** tab and click **Create Queue**.
-
- | Field | Value |
- |-------|-------|
- | Name | `Review Queue` |
- | Labels | Select the `Sentiment` label you just created |
- | Assignment Strategy | Round Robin |
- | Annotators | Add yourself |
- | Annotations Required | 1 |
-
- Click **Create** to save the queue.
-
- 
-
-
-
- Go to your **Observe** project and open the **LLM Tracing** view. Select one or more traces using the checkboxes, then click the **Add to Queue** button in the toolbar.
-
- In the dialog, choose **Review Queue** and confirm. The selected traces are now queue items with a **Pending** status.
-
-
-
- Go back to **Annotations > Queues** and click on **Review Queue** to open its detail page. Click **Start Annotating**.
-
- The annotation workspace loads the first pending item. You will see:
-
- - The trace content on the left.
- - The annotation panel on the right with your `Sentiment` label.
-
- Select an option (e.g. **Positive**), optionally add a note, and click **Submit**.
-
- 
-
- The workspace automatically advances to the next item. You can also click **Skip** to move past an item you cannot annotate.
-
-
-
- Click the **Analytics** tab on the queue detail page to see completion rates, annotator activity, and label distribution.
-
- 
-
-
-
-
-**Keyboard shortcuts** speed up annotation significantly:
-
-- **Ctrl+Enter** (or Cmd+Enter) -- Submit the current annotation
-- **1-9** -- Select a categorical option by its position
-- **S** -- Skip the current item
-
-
-## Next Steps
-
-
-
- Explore all five label types and their configuration options.
-
-
- Configure assignment strategies, multi-annotator requirements, and review workflows.
-
-
- Understand how annotation data is stored and queried via the Score model.
-
-
diff --git a/src/pages/docs/annotations/reference/label-types-and-values.mdx b/src/pages/docs/annotations/reference/label-types-and-values.mdx
new file mode 100644
index 00000000..8cf02e19
--- /dev/null
+++ b/src/pages/docs/annotations/reference/label-types-and-values.mdx
@@ -0,0 +1,81 @@
+---
+title: "Label types & values"
+description: "Settings, validation rules, and the score value shape for each label type"
+---
+
+A [label](/docs/annotations/concepts/labels) has one of five types. The type fixes the settings it needs and the control an annotator sees. Every submitted answer is written into the label's [score](/docs/annotations/concepts/scores) as JSON, in `Score.value`, and this page shows the shape that value takes for each type, plus the checks re-applied when a value is submitted. For which type to pick, see Labels; for the steps to create one, see [Create a label](/docs/annotations/guides/create-label).
+
+Every setting listed below is required by the backend; there's no default value for any of them. The prefills shown in each table are what the create drawer fills in for you, not defaults the API falls back to.
+
+
+Every type can also turn on `allow_notes`, which lets the annotator attach a free-text note alongside their answer. The note is stored in the score's `notes` field, separate from `Score.value`.
+
+
+## Categorical
+
+| Setting | What it does | Constraint | Create drawer prefills |
+|---|---|---|---|
+| `options` | The list of options the annotator picks from, each an object with a `label` field, for example `[{"label": "Good"}, {"label": "Needs work"}]` | Two or more, each non-empty, and unique once case differences are ignored | Required, no default |
+| `multi_choice` | Whether the annotator can pick more than one option | None | `false` (single choice) |
+
+Creating a categorical label also requires additional auto-annotation settings: `rule_prompt`, `auto_annotate`, and `strategy`.
+
+The annotator picks from the options you defined, one or several depending on `multi_choice`. The stored value is always an object with a `selected` array of the picked option labels: a single-select answer holds one label, for example `{"selected": ["Good"]}`; a multi-select answer holds more than one, for example `{"selected": ["Good", "Needs work"]}`.
+
+## Numeric
+
+| Setting | What it does | Constraint | Create drawer prefills |
+|---|---|---|---|
+| `min` | The lowest value on the range | 0 or greater | `0` |
+| `max` | The highest value on the range | 0 or greater, and greater than `min` | `10` |
+| `step_size` | The increment between values the annotator can land on | Greater than 0 | `1` |
+| `display_type` | Whether the annotator sees a slider or a row of buttons | `slider` or `button` | `slider` |
+
+The annotator sees a slider or a row of buttons, depending on `display_type`, stepping from `min` to `max` in `step_size` increments. The chosen value is stored as `{"value": 7.5}`.
+
+## Text
+
+| Setting | What it does | Constraint | Create drawer prefills |
+|---|---|---|---|
+| `placeholder` | Placeholder text shown in the empty field | None | `Enter your feedback...` |
+| `min_length` / `max_length` | The minimum and maximum length allowed for the submitted text | `min_length` must be less than `max_length` | `0` / `500` |
+
+The annotator gets a free-text field showing `placeholder` when empty. The entered value is stored as `{"text": "Needs a citation for the second claim"}`.
+
+## Star Rating
+
+| Setting | What it does | Constraint | Create drawer prefills |
+|---|---|---|---|
+| `no_of_stars` | How many stars the annotator sees | Greater than 0; the create drawer caps it at 10, though the backend has no upper bound | `5` |
+
+The annotator sees a row of `no_of_stars` stars to tap. The number of stars picked is stored as `{"rating": 4}`.
+
+## Thumbs Up/Down
+
+No settings beyond the type itself.
+
+The annotator sees a thumbs up / thumbs down toggle. The pick is stored as `{"value": "up"}` or `{"value": "down"}`.
+
+## Checks applied on submit
+
+Every value is checked again against the label's settings when it's submitted, not just when the label is created:
+
+- **Categorical**: every selected option must be one you defined, and if `multi_choice` is off, only one option can be selected
+- **Numeric**: the value must fall within `[min, max]`, and must land on a `step_size` increment unless it's exactly `max`
+- **Text**: the value's length must fall within `[min_length, max_length]`
+- **Star Rating**: the value must be a whole number between 1 and `no_of_stars`
+- **Thumbs Up/Down**: the value must be up or down
+
+## Keep exploring
+
+
+
+ The mental model behind types and options
+
+
+ Configure these settings on a real label
+
+
+ Where the submitted value ends up
+
+
diff --git a/src/pages/docs/annotations/reference/queue-settings-and-limits.mdx b/src/pages/docs/annotations/reference/queue-settings-and-limits.mdx
new file mode 100644
index 00000000..3deb8bf0
--- /dev/null
+++ b/src/pages/docs/annotations/reference/queue-settings-and-limits.mdx
@@ -0,0 +1,105 @@
+---
+title: "Queue settings & limits"
+description: "Every default value, permission, and cap that governs how a queue behaves"
+---
+
+This is the reference for every field, status, role, and limit that shapes a [queue](/docs/annotations/concepts/queues-and-items) in the annotation workspace. Field names below are the API/SDK payload names; where a field is also a control on the queue's form in the app, both set the same value.
+
+## Queue fields and their defaults
+
+| Field | What it holds | Default |
+|---|---|---|
+| `name` | The queue's name. Must be unique among non-archived queues in its org and scope | required, no default |
+| `description` | Free text describing the queue's purpose | empty |
+| `instructions` | Markdown guidelines shown to annotators | empty |
+| `status` | Workflow state: `draft`, `active`, `paused`, or `completed` (see transitions below) | `draft` |
+| `assignment_strategy` | How items get handed to annotators: manual, round robin, or load balanced. Round robin and load balanced are marked "Coming soon" in the app today, only manual is selectable | `manual` |
+| `annotations_required` | How many independent annotators must complete an item before it's done | `1` |
+| `reservation_timeout_minutes` | How long an opened item stays locked to the annotator who opened it before it's released back to the queue. Options are 15, 30, 60, or 240 minutes | `60` |
+| `requires_review` | Whether a completed item needs [reviewer approval](/docs/annotations/guides/review-submissions) before it counts as done. Needs an entitlement, see [Limits, caps, and gated features](#limits-caps-and-gated-features) | `false` |
+| `auto_assign` | Whether every queue member can annotate any item without being assigned to it first | `false` |
+| `is_default` | Whether this is the queue Future AGI creates automatically for a project, dataset, or agent definition | `false` |
+| `project` / `dataset` / `agent_definition` | Which one, if any, the queue is scoped to. A queue is scoped to at most one of the three, or to none for an org-level queue | none |
+
+A queue's name only has to be unique among **non-archived** queues in its scope, so archiving a queue frees up its name for reuse. The same logic caps default queues: only one **non-archived** default queue can exist per project, per dataset, and per agent definition at a time.
+
+Each [label](/docs/annotations/reference/label-types-and-values) you attach to a queue carries two settings of its own: `order`, which controls where it appears in the annotation workspace, and `required`, which forces the annotator to fill it in before submitting. Marking a label required needs an entitlement, covered in [Limits, caps, and gated features](#limits-caps-and-gated-features).
+
+## Queue statuses and the exact permitted transitions
+
+| Status | Can move to |
+|---|---|
+| Draft | Active |
+| Active | Paused, Completed |
+| Paused | Active, Completed |
+| Completed | Active, Paused |
+
+There's no path back to Draft once a queue leaves it, and Completed isn't a dead end: reopening it by moving it to Active or Paused is a normal transition, not a special case.
+
+### Archive, restore, and hard delete
+
+Archiving a queue takes it out of the active list. It stops accepting new work, but it can be restored later: everything about it (its items, labels, and annotators) comes back as it was.
+
+Hard delete is different: it's permanent. It removes the queue and everything attached to it for good, with no way to bring it back. To hard delete a queue, you have to pass `force=true` and type the queue's exact name to confirm, so it can't fire from a stray click or a typo.
+
+## Item statuses and the six source types
+
+| Status | Meaning |
+|---|---|
+| Pending | Waiting for an annotator to pick it up |
+| In Progress | An annotator has it open, reserved to them for the queue's `reservation_timeout_minutes` so nobody else can grab it in the meantime. On a queue that requires review, In Progress also covers a submitted item awaiting reviewer approval: its reservation is cleared and it's no longer open to anyone until a reviewer acts on it |
+| Completed | All required annotations have been submitted for it (and approved, if the queue requires review) |
+| Skipped | An annotator passed on it. It stays available for someone else to pick up |
+
+An item can come from six sources:
+
+| Source type | What it points to |
+|---|---|
+| Dataset row | A row from a dataset |
+| Trace | A full trace |
+| Span | A single span inside a trace |
+| Prototype | A prototype run |
+| Simulation | A simulation |
+| Session | A trace session |
+
+## Roles and what each role can do
+
+| Role | Can do |
+|---|---|
+| Annotator | Submit annotations on items in the queue |
+| Reviewer | Approve or send back submitted annotations, when the queue requires review |
+| Manager | Configure the queue: its settings, labels, and annotators |
+
+Only annotators and managers can actually submit an annotation. Holding the reviewer role by itself doesn't grant that.
+
+Whoever creates a queue becomes its first manager automatically. Org admins and workspace admins act as managers on every queue in their scope too, without ever being added as a member.
+
+## Limits, caps, and gated features
+
+| Limit | Value |
+|---|---|
+| Items per [Add Items](/docs/annotations/guides/explore-queue/add-items) call | 1,000 |
+| Filter-based selection ceiling | 10,000 items |
+| Items per synchronous [export](/docs/annotations/guides/export-annotations) | 1,000 |
+| Mentions per comment | 50 |
+| Emoji reaction length | 16 characters |
+
+
+How many queues your org can have is capped by your plan, not by the product itself, so the number depends on your plan.
+
+
+Two more things need an entitlement: turning on **Requires Review** for a queue, and marking a per-label `required` flag. Both fail with an upgrade prompt if your plan doesn't include them.
+
+## Keep exploring
+
+
+
+ The mental model behind these fields
+
+
+ Configure these settings on a real queue
+
+
+ The label settings this page doesn't cover
+
+
diff --git a/src/pages/docs/annotations/reference/sdk-api.mdx b/src/pages/docs/annotations/reference/sdk-api.mdx
new file mode 100644
index 00000000..748df193
--- /dev/null
+++ b/src/pages/docs/annotations/reference/sdk-api.mdx
@@ -0,0 +1,86 @@
+---
+title: "SDK & API"
+description: "Which surface to reach for: the dashboard, the Python SDK, or the REST API"
+---
+
+## Three ways to work with annotations
+
+The dashboard is where you set up and run a campaign: build a [queue](/docs/annotations/concepts/queues-and-items), attach [labels](/docs/annotations/concepts/labels), add annotators, and watch it through to completion.
+
+This page covers the Python SDK and the REST API. The Python SDK's `fi.queues.AnnotationQueue` client covers the queue lifecycle end to end, from a script. It:
+
+- creates queues
+- creates labels
+- adds and assigns items
+- submits annotations
+- reads progress and analytics
+- exports
+
+The REST API covers the same ground, plus every other endpoint the platform exposes. Both surfaces can also score a source directly, without a queue involved at all: a trace, span, session, dataset row, call execution, or prototype run you want to annotate without the queue workflow around it, via `create_score()` in Python or [Create Score](/docs/api/annotations/scores/create-score) over REST.
+
+## Install and authenticate
+
+```bash
+pip install futureagi
+```
+
+```python
+from fi.queues import AnnotationQueue
+
+client = AnnotationQueue(
+ fi_api_key="YOUR_API_KEY",
+ fi_secret_key="YOUR_SECRET_KEY",
+)
+```
+
+You can also set `FI_API_KEY` and `FI_SECRET_KEY` as environment variables and drop both arguments; the client picks them up automatically. Find both under **Settings → API Keys** in the platform.
+
+## An end-to-end example
+
+Creating `Support quality review`, pushing two traces into it, checking progress, then pulling the completed results back out:
+
+```python
+queue = client.create(name="Support quality review", instructions="Rate response quality 1-5")
+
+client.add_items(queue.id, items=[
+ {"source_type": "trace", "source_id": "trace_abc123"},
+ {"source_type": "trace", "source_id": "trace_def456"},
+])
+
+progress = client.get_progress(queue.id)
+print(f"{progress.completed} of {progress.total} done")
+
+results = client.export(queue.id, export_format="json", status="completed")
+```
+
+## Job to method to endpoint
+
+Each job below has a Python method and a REST endpoint that do the same thing. Full parameter tables live on the linked SDK pages, not here.
+
+| Job | Python SDK | REST API |
+|---|---|---|
+| Create a label | [`create_label()`](/docs/sdk/annotation-queues/labels) | [Create Label](/docs/api/annotations/labels/create-label) |
+| Create a queue | [`create()`](/docs/sdk/annotation-queues/queues) | [Create Queue](/docs/api/annotations/queues/create-queue) |
+| Add items | [`add_items()`](/docs/sdk/annotation-queues/items) | [Add Items](/docs/api/annotations/items/add-items) |
+| Submit annotations for a queue item | [`submit_annotations()`](/docs/sdk/annotation-queues/annotations) | [Submit Annotations](/docs/api/annotations/items/submit-annotations) |
+| Score a source directly | [`create_score()`](/docs/sdk/annotation-queues/scores) | [Create Score](/docs/api/annotations/scores/create-score) |
+| Read progress | [`get_progress()`](/docs/sdk/annotation-queues/analytics) | [Get Progress](/docs/api/annotations/queues/get-progress) |
+| Export | [`export()`](/docs/sdk/annotation-queues/export) | [Export](/docs/api/annotations/queues/export) |
+
+
+The dashboard's caps apply to the SDK and the REST API too, not just the UI: up to 1,000 items per `add_items()` call, and up to 1,000 items per synchronous `export()` call. Go over either and the call errors instead of hanging. See [Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits).
+
+
+## Keep exploring
+
+
+
+ The full method reference, one page per concept
+
+
+ Every field, status, role, and cap a queue runs under
+
+
+ Every REST endpoint across the platform
+
+
diff --git a/src/pages/docs/annotations/sdk/annotation-queue-using-sdk.mdx b/src/pages/docs/annotations/sdk/annotation-queue-using-sdk.mdx
deleted file mode 100644
index 5b84537b..00000000
--- a/src/pages/docs/annotations/sdk/annotation-queue-using-sdk.mdx
+++ /dev/null
@@ -1,448 +0,0 @@
----
-title: "Annotation Queues via Python SDK"
-description: "Create queues, manage labels, add items, submit annotations, track progress, and export results programmatically with the Future AGI Python SDK."
----
-
-Annotation queues let you organize traces, sessions, datasets, and simulation outputs for structured human review. Using the SDK, you can:
-
-- Create and configure annotation queues programmatically
-- Create and manage annotation labels (categorical, text, numeric, star, thumbs up/down)
-- Add items from multiple sources (traces, spans, sessions, dataset rows, simulations, prototype runs)
-- Submit or import annotations in bulk
-- Track progress and inter-annotator agreement
-- Export annotated data to datasets
-
-
-All methods that accept `queue_id` also accept `queue_name` as an alternative. Similarly, methods that accept `label_id` also accept `label_name`. The SDK resolves names to IDs automatically.
-
-
----
-
-## Installation
-
-```bash
-pip install futureagi
-```
-
-## Authentication
-
-You can find your API key and secret key under **Build > Keys** in the sidebar.
-
-
-
-
-```python
-from fi.queues import AnnotationQueue
-
-client = AnnotationQueue(
- fi_api_key="YOUR_API_KEY",
- fi_secret_key="YOUR_SECRET_KEY",
-)
-```
-
----
-
-## Creating Labels
-
-Create annotation labels to define what annotators should evaluate. Each label has a type that determines the kind of input annotators provide:
-
-```python
-# Categorical label (multiple choice)
-sentiment_label = client.create_label(
- name="Sentiment",
- type="categorical",
- settings={
- "rule_prompt": "Classify the sentiment of the response",
- "multi_choice": False,
- "options": [
- {"label": "Positive"},
- {"label": "Negative"},
- {"label": "Neutral"},
- ],
- "auto_annotate": False,
- "strategy": None,
- },
-)
-
-# Numeric label (slider or buttons)
-quality_label = client.create_label(
- name="Quality Score",
- type="numeric",
- settings={
- "min": 1,
- "max": 10,
- "step_size": 1,
- "display_type": "slider",
- },
-)
-
-# Thumbs up/down label
-thumbs_label = client.create_label(
- name="Helpful",
- type="thumbs_up_down",
- settings={},
-)
-```
-
-You can also list and retrieve existing labels by ID or name:
-
-```python
-# List all labels
-labels = client.list_labels()
-
-# List labels scoped to a project
-project_labels = client.list_labels(project_id="your_project_id")
-
-# Get a specific label by ID or name
-label = client.get_label(label_id="your_label_id")
-label = client.get_label(label_name="Sentiment")
-
-# Delete a label by ID or name
-client.delete_label(label_id="your_label_id")
-client.delete_label(label_name="Sentiment")
-```
-
----
-
-## Creating an Annotation Queue
-
-You can find all your annotation queues under **Observe > Annotations** in the sidebar.
-
-
-
-
-Create a queue with instructions and configuration for your reviewers:
-
-```python
-queue = client.create(
- name="Sentiment Review",
- description="Review and label sentiment of customer interactions",
- instructions="Rate each trace as positive, negative, or neutral. Consider the overall tone of the conversation.",
- assignment_strategy="load_balanced",
- annotations_required=2,
- reservation_timeout_minutes=60,
- requires_review=True,
-)
-print(f"Created queue: {queue.id} (status: {queue.status})")
-```
-
-**Assignment strategies:**
-- `"manual"` — Explicitly assign items to annotators
-- `"round_robin"` — Distribute items evenly across annotators
-- `"load_balanced"` — Assign to the annotator with the fewest pending items
-
----
-
-## Attaching Labels to a Queue
-
-Once labels are created, attach them to a queue using IDs or names:
-
-```python
-# Using IDs (from create_label return values)
-client.add_label(queue.id, label_id=sentiment_label.id)
-client.add_label(queue.id, label_id=quality_label.id)
-
-# Or using names
-client.add_label(queue_name="Sentiment Review", label_name="Sentiment")
-client.add_label(queue_name="Sentiment Review", label_name="Quality Score")
-```
-
----
-
-## Activating the Queue
-
-Click on a queue to view its settings, including the queue name, description, instructions, and attached labels.
-
-
-
-
-Queues start in `draft` status. Activate when ready for annotation:
-
-```python
-# Using ID
-queue = client.activate(queue.id)
-
-# Or using name
-queue = client.activate(queue_name="Sentiment Review")
-print(f"Queue status: {queue.status}") # "active"
-```
-
----
-
-## Adding Items
-
-Add items from various sources to the queue:
-
-```python
-result = client.add_items(queue.id, items=[
- {"source_type": "trace", "source_id": "trace_uuid_1"},
- {"source_type": "trace", "source_id": "trace_uuid_2"},
- {"source_type": "observation_span", "source_id": "span_uuid_1"},
- {"source_type": "dataset_row", "source_id": "row_uuid_1"},
- {"source_type": "trace_session", "source_id": "session_uuid_1"},
- {"source_type": "call_execution", "source_id": "simulation_uuid_1"},
- {"source_type": "prototype_run", "source_id": "prototype_run_uuid_1"},
-])
-print(f"Added: {result.added}, Duplicates: {result.duplicates}")
-```
-
-**Supported source types:** `trace`, `observation_span`, `trace_session`, `call_execution`, `prototype_run`, `dataset_row`
-
----
-
-## Listing and Filtering Items
-
-```python
-# List all pending items
-pending_items = client.list_items(queue.id, status="pending")
-
-# List items assigned to a specific user
-assigned_items = client.list_items(queue.id, assigned_to="user_uuid")
-
-# Paginate through items
-page_2 = client.list_items(queue.id, page=2, page_size=20)
-```
-
----
-
-## Assigning Items
-
-Manually assign items to annotators:
-
-```python
-# Assign items to a user
-client.assign_items(
- queue.id,
- item_ids=[items[0].id, items[1].id],
- user_id="annotator_user_id",
-)
-
-# Unassign items
-client.assign_items(
- queue.id,
- item_ids=[items[0].id],
- user_id=None,
-)
-```
-
----
-
-## Submitting Annotations
-
-In the UI, annotators see each item's content alongside the configured labels and can submit their annotations directly.
-
-
-
-
-Submit annotations as the authenticated user:
-
-```python
-client.submit_annotations(
- queue.id,
- item_id=items[0].id,
- annotations=[
- {"label_id": "sentiment_label_id", "value": "positive"},
- {"label_id": "confidence_label_id", "value": 0.95},
- ],
- notes="Clear positive sentiment throughout the conversation",
-)
-```
-
----
-
-## Importing Annotations Programmatically
-
-Import annotations from an external source or automated pipeline:
-
-```python
-result = client.import_annotations(
- queue.id,
- item_id=items[0].id,
- annotations=[
- {"label_id": "sentiment_label_id", "value": "positive"},
- {"label_id": "confidence_label_id", "value": 0.92},
- ],
- annotator_id="external_annotator_user_id", # optional
-)
-print(f"Imported: {result.imported}")
-```
-
----
-
-## Completing and Skipping Items
-
-```python
-# Mark item as completed
-client.complete_item(queue.id, item_id=items[0].id)
-
-# Skip an item
-client.skip_item(queue.id, item_id=items[1].id)
-```
-
----
-
-## Tracking Progress
-
-```python
-progress = client.get_progress(queue.id)
-print(f"Total: {progress.total}")
-print(f"Completed: {progress.completed}")
-print(f"Pending: {progress.pending}")
-print(f"Progress: {progress.progress_pct}%")
-```
-
----
-
-## Analytics and Agreement
-
-The Analytics tab shows throughput, status breakdown, label distribution, and annotator performance.
-
-
-
-
-```python
-# Get throughput and annotator performance
-analytics = client.get_analytics(queue.id)
-print(f"Status breakdown: {analytics.status_breakdown}")
-print(f"Total completed: {analytics.throughput['total_completed']}")
-print(f"Avg per day: {analytics.throughput['avg_per_day']}")
-
-# Daily throughput (last 30 days)
-for day in analytics.throughput["daily"]:
- print(f" {day['date']}: {day['count']} completed")
-
-# Get inter-annotator agreement
-agreement = client.get_agreement(queue.id)
-print(f"Overall agreement: {agreement.overall_agreement}")
-```
-
----
-
-## Exporting Results
-
-### Export as JSON or CSV
-
-```python
-# Export completed annotations as JSON
-data = client.export(queue.id, export_format="json", status="completed")
-
-# Export as CSV
-csv_data = client.export(queue.id, export_format="csv", status="completed")
-```
-
-### Export to a Dataset
-
-```python
-# Create a new dataset from annotations
-result = client.export_to_dataset(queue.id, dataset_name="Sentiment Labels")
-print(f"Created dataset '{result.dataset_name}' with {result.rows_created} rows")
-
-# Or append to an existing dataset
-result = client.export_to_dataset(queue.id, dataset_id="existing_dataset_uuid")
-```
-
----
-
-## Using Scores Without a Queue
-
-You can also annotate any source entity directly using scores, without creating a queue:
-
-```python
-# Create a single score (by label ID or name)
-score = client.create_score(
- source_type="trace",
- source_id="trace_uuid_1",
- label_name="Quality Score",
- value="good",
- score_source="api",
- notes="Automated quality check",
-)
-
-# Create multiple scores at once
-client.create_scores(
- source_type="trace",
- source_id="trace_uuid_1",
- scores=[
- {"label_id": "quality_label_id", "value": "good"},
- {"label_id": "relevance_label_id", "value": 4.5},
- ],
-)
-
-# Retrieve scores
-scores = client.get_scores(source_type="trace", source_id="trace_uuid_1")
-for s in scores:
- print(f"{s.label_name}: {s.value} (by {s.annotator_name})")
-```
-
----
-
-## Completing a Queue
-
-When all items have been reviewed:
-
-```python
-queue = client.complete_queue(queue.id)
-print(f"Queue status: {queue.status}") # "completed"
-```
-
-
-Completing a queue does **not** automatically disable its automation rules. If you have active rules, they may continue adding items, which will re-activate the queue. Disable or delete automation rules manually before completing the queue.
-
-
----
-
-## Complete Example
-
-```python
-from fi.queues import AnnotationQueue
-
-client = AnnotationQueue(
- fi_api_key="YOUR_API_KEY",
- fi_secret_key="YOUR_SECRET_KEY",
-)
-
-# 1. Create and configure the queue
-queue = client.create(
- name="Trace Quality Review",
- instructions="Rate the quality of each AI response on a scale of 1-5",
- assignment_strategy="round_robin",
- annotations_required=2,
-)
-
-# 2. Create a label, attach it, and activate
-label = client.create_label(
- name="Quality",
- type="numeric",
- settings={"min": 1, "max": 5, "step_size": 1, "display_type": "slider"},
-)
-client.add_label(queue.id, label.id)
-queue = client.activate(queue.id)
-
-# 3. Add items (using queue name works too)
-result = client.add_items(queue_name="Trace Quality Review", items=[
- {"source_type": "trace", "source_id": "trace_1"},
- {"source_type": "trace", "source_id": "trace_2"},
- {"source_type": "trace", "source_id": "trace_3"},
-])
-print(f"Added {result.added} items")
-
-# 4. List and annotate items
-items = client.list_items(queue.id, status="pending")
-for item in items:
- client.submit_annotations(
- queue.id,
- item.id,
- annotations=[{"label_id": label.id, "value": 4}],
- )
- client.complete_item(queue.id, item.id)
-
-# 5. Check progress and export
-progress = client.get_progress(queue_name="Trace Quality Review")
-print(f"Completed: {progress.completed}/{progress.total}")
-
-export_result = client.export_to_dataset(queue.id, dataset_name="Quality Reviews")
-print(f"Exported to dataset: {export_result.dataset_name}")
-
-# 6. Complete the queue
-client.complete_queue(queue.id)
-```
diff --git a/src/pages/docs/annotations/sdk/javascript.mdx b/src/pages/docs/annotations/sdk/javascript.mdx
deleted file mode 100644
index 61e2bc16..00000000
--- a/src/pages/docs/annotations/sdk/javascript.mdx
+++ /dev/null
@@ -1,303 +0,0 @@
----
-title: "Annotations JavaScript & TypeScript SDK"
-description: "Log annotations, manage queues, submit scores, and export results using the FutureAGI JavaScript/TypeScript SDK's Annotation and AnnotationQueue classes."
----
-
-# JavaScript SDK
-
-The FutureAGI JavaScript/TypeScript SDK provides two primary classes: `Annotation` for logging annotations via a DataFrame-style interface, and `AnnotationQueue` for full queue lifecycle management.
-
-## Installation
-
-
-
-```bash npm
-npm install @future-agi/sdk
-```
-
-```bash yarn
-yarn add @future-agi/sdk
-```
-
-```bash pnpm
-pnpm add @future-agi/sdk
-```
-
-
-
----
-
-## Annotation Class -- Log Annotations
-
-### Initialize the client
-
-```typescript
-import { Annotation } from '@future-agi/sdk';
-
-const client = new Annotation({
- fiApiKey: 'YOUR_API_KEY',
- fiSecretKey: 'YOUR_SECRET_KEY',
-});
-```
-
-### Log annotations
-
-Log annotations using DataFrame-style records. Each record is an object with column keys following the same naming convention as the [Python SDK](/docs/annotations/sdk/python).
-
-```typescript
-const response = await client.logAnnotations([
- {
- 'context.span_id': 'span_abc123',
- 'annotation.quality.text': 'Excellent response',
- 'annotation.sentiment.label': 'positive',
- 'annotation.accuracy.score': 9.0,
- 'annotation.rating.rating': 5,
- 'annotation.helpful.thumbs': true,
- 'annotation.notes': 'Top quality',
- },
- {
- 'context.span_id': 'span_def456',
- 'annotation.quality.text': 'Needs improvement',
- 'annotation.sentiment.label': 'negative',
- 'annotation.accuracy.score': 3.5,
- 'annotation.rating.rating': 2,
- 'annotation.helpful.thumbs': false,
- 'annotation.notes': 'Hallucinated facts',
- },
-], { projectName: 'My Project' });
-
-console.log(`Created: ${response.annotationsCreated}, Errors: ${response.errorsCount}`);
-```
-
-
-For the full column naming convention table, see the [Python SDK -- Column naming convention](/docs/annotations/sdk/python#column-naming-convention). The format is identical across both SDKs.
-
-
-### Get labels
-
-```typescript
-const labels = await client.getLabels({ projectId: 'proj_123' });
-
-labels.forEach(l => console.log(`${l.name} (${l.type}): ${l.id}`));
-```
-
-### List projects
-
-```typescript
-const projects = await client.listProjects({ projectType: 'observe' });
-
-projects.forEach(p => console.log(`${p.name}: ${p.id}`));
-```
-
----
-
-## AnnotationQueue Class -- Full Queue Management
-
-The `AnnotationQueue` class provides complete programmatic control over the annotation queue lifecycle: creating queues, adding items, assigning work, submitting annotations, and exporting results.
-
-### Initialize the client
-
-```typescript
-import { AnnotationQueue } from '@future-agi/sdk';
-
-const queues = new AnnotationQueue({
- fiApiKey: 'YOUR_API_KEY',
- fiSecretKey: 'YOUR_SECRET_KEY',
-});
-```
-
-### Create a queue
-
-```typescript
-const queue = await queues.create({
- name: 'Review Queue',
- description: 'Quality review of traces',
- instructions: 'Rate response quality on all labels',
- assignmentStrategy: 'round_robin',
- annotationsRequired: 2,
- reservationTimeoutMinutes: 30,
- requiresReview: false,
-});
-```
-
-### Add items to a queue
-
-```typescript
-const result = await queues.addItems(queue.id, [
- { sourceType: 'trace', sourceId: 'trace_abc' },
- { sourceType: 'observation_span', sourceId: 'span_def' },
- { sourceType: 'dataset_row', sourceId: 'row_ghi' },
-]);
-
-console.log(`Added: ${result.added}, Duplicates: ${result.duplicates}`);
-```
-
-#### Valid source types
-
-| Source Type | Description |
-|-------------|-------------|
-| `trace` | An LLM trace |
-| `observation_span` | A specific span in a trace |
-| `trace_session` | A conversation session |
-| `dataset_row` | A dataset row |
-| `call_execution` | A simulation call |
-| `prototype_run` | A prototype run |
-
-### Submit annotations
-
-```typescript
-await queues.submitAnnotations(queue.id, itemId, [
- { labelId: 'label_123', value: 'positive', scoreSource: 'human' },
- { labelId: 'label_456', value: 4.5, scoreSource: 'human' },
-], { notes: 'High quality response' });
-```
-
-### Create scores directly (without queue)
-
-You can create scores against any source without going through a queue workflow.
-
-```typescript
-const score = await queues.createScore({
- sourceType: 'trace',
- sourceId: 'trace_abc',
- labelId: 'label_123',
- value: { text: 'Good response' },
- scoreSource: 'human',
- notes: 'Quick feedback',
-});
-```
-
-### Bulk create scores
-
-```typescript
-await queues.createScores({
- sourceType: 'trace',
- sourceId: 'trace_abc',
- scores: [
- { labelId: 'label_123', value: 'positive' },
- { labelId: 'label_456', value: 4.5 },
- ],
- notes: 'Batch annotation',
-});
-```
-
-### Queue lifecycle
-
-```typescript
-// Activate a draft queue
-await queues.activate(queue.id);
-
-// Mark a queue as completed
-await queues.completeQueue(queue.id);
-
-// Add or remove labels from a queue
-await queues.addLabel(queue.id, 'label_789');
-await queues.removeLabel(queue.id, 'label_789');
-
-// List items with optional status filter
-const items = await queues.listItems(queue.id, { status: 'pending' });
-
-// Assign items to a specific user
-await queues.assignItems(queue.id, ['item_1', 'item_2'], 'user_123');
-
-// Complete or skip items
-await queues.completeItem(queue.id, 'item_1');
-await queues.skipItem(queue.id, 'item_2');
-```
-
-### Progress and analytics
-
-```typescript
-const progress = await queues.getProgress(queue.id);
-console.log(`${progress.completed}/${progress.total} (${progress.progressPct}%)`);
-
-const analytics = await queues.getAnalytics(queue.id);
-
-const agreement = await queues.getAgreement(queue.id);
-```
-
-### Export
-
-
-
-```typescript JSON export
-const data = await queues.export(queue.id, {
- format: 'json',
- status: 'completed',
-});
-```
-
-```typescript Export to dataset
-const dataset = await queues.exportToDataset(queue.id, {
- datasetName: 'Annotated Traces Q1',
- statusFilter: 'completed',
-});
-
-console.log(`Created dataset ${dataset.datasetId} with ${dataset.rowsCreated} rows`);
-```
-
-
-
----
-
-## Complete Method Reference
-
-### AnnotationQueue methods
-
-| Method | Description |
-|--------|-------------|
-| `create(config)` | Create a new queue |
-| `list(options)` | List queues |
-| `get(queueId)` | Get queue details |
-| `update(queueId, updates)` | Update queue configuration |
-| `delete(queueId)` | Delete a queue |
-| `activate(queueId)` | Set queue status to active |
-| `completeQueue(queueId)` | Set queue status to completed |
-| `addLabel(queueId, labelId)` | Add a label to a queue |
-| `removeLabel(queueId, labelId)` | Remove a label from a queue |
-| `addItems(queueId, items)` | Add source items to a queue |
-| `listItems(queueId, options)` | List queue items with optional filters |
-| `removeItems(queueId, itemIds)` | Remove items from a queue |
-| `assignItems(queueId, itemIds, userId)` | Assign items to a user |
-| `submitAnnotations(queueId, itemId, annotations)` | Submit annotations for an item |
-| `getAnnotations(queueId, itemId)` | Get annotations for an item |
-| `completeItem(queueId, itemId)` | Mark an item as completed |
-| `skipItem(queueId, itemId)` | Skip an item |
-| `createScore(options)` | Create a single score (no queue required) |
-| `createScores(options)` | Bulk create scores (no queue required) |
-| `getScores(sourceType, sourceId)` | Get scores for a source |
-| `getProgress(queueId)` | Get queue completion progress |
-| `getAnalytics(queueId)` | Get queue analytics and metrics |
-| `getAgreement(queueId)` | Get inter-annotator agreement metrics |
-| `export(queueId, options)` | Export annotations as JSON or CSV |
-| `exportToDataset(queueId, options)` | Export annotations to a FutureAGI dataset |
-
----
-
-## Best Practices
-
-- **Use `logAnnotations()` for bulk SDK-based annotation** -- The DataFrame-style format is the fastest way to annotate many spans at once.
-- **Use `AnnotationQueue` for programmatic queue management** -- Create, assign, and complete queues entirely from code.
-- **Use `createScore()` / `createScores()` for direct score creation** -- Bypass the queue workflow when you need to attach scores to traces directly.
-- **Always handle errors** -- Check for partial failures in bulk operations. Both `logAnnotations` and `addItems` can succeed for some records and fail for others.
-- **Use TypeScript** -- All SDK methods are fully typed. TypeScript catches column name typos and invalid configurations at compile time.
-
-
-Bulk operations (`logAnnotations`, `addItems`, `createScores`) may partially succeed. Always inspect the response for per-record errors before assuming all records were processed.
-
-
----
-
-## Next steps
-
-
-
- DataFrame-based annotation logging with the Python SDK.
-
-
- Query and manage annotation scores via the REST API.
-
-
- REST API reference for queue CRUD operations.
-
-
diff --git a/src/pages/docs/annotations/sdk/python.mdx b/src/pages/docs/annotations/sdk/python.mdx
deleted file mode 100644
index 37745faa..00000000
--- a/src/pages/docs/annotations/sdk/python.mdx
+++ /dev/null
@@ -1,160 +0,0 @@
----
-title: "Annotations Python SDK: Log & Manage"
-description: "Log annotations via DataFrame, retrieve labels, list projects, and submit human feedback to traces using the FutureAGI Python SDK."
----
-
-# Python SDK
-
-The FutureAGI Python SDK provides a simple, DataFrame-based interface for logging annotations against your traces. Install the package, authenticate, and start annotating in minutes.
-
-## Installation
-
-
-
-```bash pip
-pip install futureagi
-```
-
-```bash pip3
-pip3 install futureagi
-```
-
-
-
-## Authentication
-
-```python
-from fi.annotations import Annotation
-
-client = Annotation(
- fi_api_key="YOUR_API_KEY",
- fi_secret_key="YOUR_SECRET_KEY",
-)
-```
-
-
-You can also set `FI_API_KEY` and `FI_SECRET_KEY` as environment variables. The client picks them up automatically when no arguments are passed.
-
-
----
-
-## Log Annotations
-
-The `log_annotations()` method accepts a pandas DataFrame where each row represents one annotation record. Columns follow the naming convention `annotation..`.
-
-### Column naming convention
-
-| Column Pattern | Label Type | Example Value |
-|----------------|------------|---------------|
-| `annotation..text` | Text | `"good response"` |
-| `annotation..label` | Categorical | `"positive"` |
-| `annotation..score` | Numeric | `8.5` |
-| `annotation..rating` | Star (1-5) | `4` |
-| `annotation..thumbs` | Thumbs Up/Down | `True` |
-| `annotation.notes` | Notes (shared) | `"Great response!"` |
-| `context.span_id` | (required) Span ID | `"span_abc123"` |
-
-
-Every row **must** include a `context.span_id` column. This links the annotation to a specific span in your Observe project.
-
-
-### Full example
-
-```python
-import pandas as pd
-from fi.annotations import Annotation
-
-client = Annotation(
- fi_api_key="YOUR_API_KEY",
- fi_secret_key="YOUR_SECRET_KEY",
-)
-
-df = pd.DataFrame({
- "context.span_id": ["span_abc123", "span_def456"],
- "annotation.quality.text": ["Excellent response", "Needs improvement"],
- "annotation.sentiment.label": ["positive", "negative"],
- "annotation.accuracy.score": [9.0, 3.5],
- "annotation.rating.rating": [5, 2],
- "annotation.helpful.thumbs": [True, False],
- "annotation.notes": ["Top quality", "Hallucinated facts"],
-})
-
-response = client.log_annotations(df, project_name="My Project")
-print(f"Created: {response.annotations_created}, Errors: {response.errors_count}")
-```
-
-### Response object
-
-| Field | Type | Description |
-|-------|------|-------------|
-| `message` | `str` | Summary message |
-| `annotations_created` | `int` | New annotations created |
-| `annotations_updated` | `int` | Existing annotations updated |
-| `notes_created` | `int` | Notes created |
-| `succeeded_count` | `int` | Successful records |
-| `errors_count` | `int` | Failed records |
-| `errors` | `list` | Error details per failed record |
-
----
-
-## Get Labels
-
-Retrieve all annotation labels configured for a project. Use the returned label IDs when constructing your DataFrame columns.
-
-```python
-labels = client.get_labels(project_id="proj_123")
-
-for label in labels:
- print(f"{label.name} ({label.type}): {label.id}")
-```
-
----
-
-## List Projects
-
-List all projects accessible to your API key. Filter by project type to find your Observe projects.
-
-```python
-projects = client.list_projects(project_type="observe")
-
-for p in projects:
- print(f"{p.name}: {p.id}")
-```
-
----
-
-## Annotation Queues
-
-
-For queue management -- creating queues, adding items, submitting annotations, and exporting results -- use the REST API directly or the [JavaScript SDK](/docs/annotations/sdk/javascript) which provides full queue support. See the [Queues API reference](/docs/api/annotations/queues/create-queue) for details.
-
-
----
-
-## Best Practices
-
-- **Batch annotations** -- Group 100--500 records per DataFrame for optimal throughput.
-- **Consistent span IDs** -- Ensure span IDs match traces in your Observe project. Invalid IDs result in per-row errors.
-- **Idempotent notes** -- Duplicate notes for the same span are silently skipped.
-- **Error handling** -- Always check `response.errors_count` and inspect `response.errors` for partial failures.
-- **Label IDs** -- Use `get_labels()` to fetch label names and IDs before constructing your DataFrame.
-
-
-Annotations are immutable once submitted. Double-check your DataFrame before calling `log_annotations()`.
-
-
----
-
-## Next steps
-
-
-
- Full queue management, scores, and annotation support in JavaScript/TypeScript.
-
-
- Query and manage annotation scores via the REST API.
-
-
- Upload annotations in bulk using the REST API directly.
-
-
diff --git a/src/pages/docs/annotations/troubleshooting.mdx b/src/pages/docs/annotations/troubleshooting.mdx
new file mode 100644
index 00000000..1d16e0d0
--- /dev/null
+++ b/src/pages/docs/annotations/troubleshooting.mdx
@@ -0,0 +1,71 @@
+---
+title: "Annotation FAQ & fixes"
+description: "Common annotation questions, and fixes for the errors you hit most"
+---
+
+## In this page
+
+The questions people ask most about annotation, and the errors they run into, with a direct fix for each. Hit an error? Jump straight to [Common errors and fixes](#common-errors-and-fixes). If your answer isn't here, reach out via [support](https://futureagi.com/contact-us).
+
+## Common errors and fixes
+
+| Symptom | Cause | Fix |
+|---|---|---|
+| The submit button won't enable | Every label attached to the queue needs an answer before you can submit | Answer every label, submit stays disabled until all have values; pressing Ctrl+Enter while any are empty lists which ones are still open |
+| An item says it's reserved by someone else | Another annotator already has the item open | Skip to Next Item, or come back once the other annotator submits or skips it |
+| You can't annotate because the queue isn't active | The queue is in draft or paused | Ask a queue manager to switch it to active from the Settings tab |
+| An item is assigned to someone else | Auto-assign is off, and the item was assigned to another annotator | Ask a queue manager to reassign it to you |
+| Skip is refused on an item | The item is already completed, or it's pending review on a queue that requires review | Completed items can't be skipped; an item waiting on review has to clear review first |
+| An Add Items call is rejected | The payload has more than 1,000 items, so the API returns HTTP 413 | Split the items into batches of 1,000 or fewer, see [Add items](/docs/annotations/guides/explore-queue/add-items) |
+| A filter-mode selection is rejected | The filter resolves to more than 10,000 items | Narrow the filter, or add items in smaller batches |
+| An export of a large queue fails | A synchronous export refuses queues with more than 1,000 items outright rather than truncating them, returning HTTP 413 | Narrow the filter so it resolves to 1,000 items or fewer, see [Export annotations](/docs/annotations/guides/export-annotations) |
+| A numeric or text value is rejected on submit | The value falls outside the label's configured min, max, step size, or length | Match the value to the label's settings, see [Label types & values](/docs/annotations/reference/label-types-and-values) |
+| A queue name is rejected as already taken | Another queue in the same scope already uses that name | Choose a different name |
+| A hard delete refuses to go through | Hard delete needs the queue's exact name typed as confirmation, plus a force flag on the API | Type the queue's exact name to confirm, the Delete forever button in the dialog stays disabled until it matches; via the API, pass `force=true` with the exact name |
+
+## Roles and permissions
+
+**Who can annotate, and who can review?**
+
+Annotating a queue needs the annotator or manager role on it; without one of those roles, submitting is refused. Org and workspace admins get manager-level access automatically, without being added to the queue explicitly. Reviewing has its own role, see [Review submissions](/docs/annotations/guides/review-submissions) for how it works, and [Queue settings & limits](/docs/annotations/reference/queue-settings-and-limits) for the full roles table.
+
+**Why don't I see the Settings or Rules tab?**
+
+Both are manager-only surfaces. If you're not a manager on the queue, and not an org or workspace admin, they stay hidden.
+
+## Scores
+
+**Why does the same trace show two scores from the same person?**
+
+Score the same trace from two different queues and you get two independent [scores](/docs/annotations/concepts/scores), not one overwritten value.
+
+**If I edit a score, do I lose the old value?**
+
+No. Changing a score's value appends to its history instead of overwriting it. Previous values show in the annotation history panel on the item, listed as Previous 1, Previous 2, and so on.
+
+## Queues
+
+**Why is the Agreement tab empty?**
+
+Agreement measures how much annotators agree, so it has nothing to compare until more than one independent submission lands on the same items. See [Track progress & agreement](/docs/annotations/guides/explore-queue/progress-and-agreement) for what the queue needs to populate it.
+
+**What happens to items when a queue is archived?**
+
+The items stay in the queue. Archiving switches the queues list to Archived, and any rules attached to it pause. Restore it from the Archived view and it comes back in the status it had when you archived it.
+
+## Keep exploring
+
+
+
+ The operational model behind a queue and its items
+
+
+ Why a score outlives the queue item that created it
+
+
+ The settings and validation rules behind every label type
+
+
+ Fields, statuses, roles, and the hard caps on a queue
+
+
diff --git a/src/pages/docs/cookbook/decrease-hallucination.mdx b/src/pages/docs/cookbook/decrease-hallucination.mdx
index 3727b475..8d3fe329 100644
--- a/src/pages/docs/cookbook/decrease-hallucination.mdx
+++ b/src/pages/docs/cookbook/decrease-hallucination.mdx
@@ -238,7 +238,7 @@ To quantify performance of each combination of RAG setup, a set of evals accordi
- **`criteria`**: Description of the criteria for evaluation
- Returns a percentage score, where a high-score Indicate that the context is relevant or sufficient to produce an accurate and coherent output.
- Click [here](/docs/prototype/features/evals) to learn more about the evals provided by Future AGI
+ Click [here](/docs/evaluation/builtin) to learn more about the evals provided by Future AGI
The `eval_tags` list contains multiple instances of `EvalTag`. Each `EvalTag` represents a specific evaluation configuration to be applied during runtime, encapsulating all necessary parameters for the evaluation process.
@@ -253,13 +253,13 @@ Parameters of `EvalTag` :
- For Context Adherence Eval, `EvalName.CONTEXT_ADHERENCE`,
- For Context Retrieval Quality,`EvalName.EVAL_CONTEXT_RETRIEVAL_QUALITY`
- Click [here](/docs/prototype/features/evals) to get complete list of evals provided by Future AGI
+ Click [here](/docs/evaluation/builtin) to get complete list of evals provided by Future AGI
- **`config`**: Dictionary for providing specific configurations for the evaluation. An empty dictionary `{}` means that default configuration parameters will be used.
- Click [here](/docs/prototype/features/evals) to learn more about what config is required for corresponding evals
+ Click [here](/docs/evaluation/builtin) to learn more about what config is required for corresponding evals
- **`mapping`**: This dictionary maps the required inputs for the evaluation to specific attributes of the operation.
- Click [here](/docs/prototype/features/evals) to learn more about what inputs are required for corresponding evals
+ Click [here](/docs/evaluation/builtin) to learn more about what inputs are required for corresponding evals
- **`custom_eval_name`**: A user-defined name for the specific evaluation instance.
**7.2 Setting Up Trace Provider**
diff --git a/src/pages/docs/cookbook/error-feed/google-adk-multi-agent.mdx b/src/pages/docs/cookbook/error-feed/google-adk-multi-agent.mdx
index e3d64954..cc4affb0 100644
--- a/src/pages/docs/cookbook/error-feed/google-adk-multi-agent.mdx
+++ b/src/pages/docs/cookbook/error-feed/google-adk-multi-agent.mdx
@@ -7,7 +7,7 @@ description: "Build a Google ADK multi-agent system with tracing, then use Futur
This cookbook walks through a complete example: build a multi-agent system with Google ADK, instrument it with tracing, and use [Error Feed](/docs/error-feed) to automatically analyze agent performance. By the end, you'll have traces flowing into Observe with Error Feed scores and recommendations visible on each trace.
-For a framework-agnostic guide on reading Error Feed results, see [Issue Overview](/docs/error-feed/features/issue-overview).
+For a framework-agnostic guide on reading Error Feed results, see [Investigate an issue](/docs/error-feed/guides/investigate-an-issue).
---
@@ -236,12 +236,12 @@ Click on a trace to open the trace tree. Error Feed insights appear in a collaps

-For details on how to read scores, insights, clusters, and recommendations, see [Issue Overview](/docs/error-feed/features/issue-overview).
+For details on how to read scores, insights, clusters, and recommendations, see [Investigate an issue](/docs/error-feed/guides/investigate-an-issue).
---
## Next Steps
-- [Issue Overview](/docs/error-feed/features/issue-overview): Understand scores, clusters, and recommendations
-- [Error Taxonomy](/docs/error-feed/concepts/taxonomy): Explore all error categories
+- [Investigate an issue](/docs/error-feed/guides/investigate-an-issue): Understand scores, clusters, and recommendations
+- [Error taxonomy](/docs/error-feed/reference/error-taxonomy): Explore all error categories
- [Set Up Observability](/docs/quickstart/setup-observability): Send traces from other frameworks
diff --git a/src/pages/docs/cookbook/eval-metrics-optimization.mdx b/src/pages/docs/cookbook/eval-metrics-optimization.mdx
index f5eede53..400c7eae 100644
--- a/src/pages/docs/cookbook/eval-metrics-optimization.mdx
+++ b/src/pages/docs/cookbook/eval-metrics-optimization.mdx
@@ -165,7 +165,7 @@ data_mapper = BasicDataMapper(key_map={"response": "generated_output"})
See a complete end-to-end example of running an optimization.
diff --git a/src/pages/docs/cookbook/falcon-ai/end-to-end.mdx b/src/pages/docs/cookbook/falcon-ai/end-to-end.mdx
index 1530db76..eb4c0ba2 100644
--- a/src/pages/docs/cookbook/falcon-ai/end-to-end.mdx
+++ b/src/pages/docs/cookbook/falcon-ai/end-to-end.mdx
@@ -191,7 +191,7 @@ You went from a noisy traced project to a fixed agent and a reusable regression
Curate balanced golden datasets from real traces with `/build-dataset`
-
+
All built-in slash commands and how to write your own
diff --git a/src/pages/docs/cookbook/falcon-ai/eval-datasets-from-traces.mdx b/src/pages/docs/cookbook/falcon-ai/eval-datasets-from-traces.mdx
index 81735884..de8f7104 100644
--- a/src/pages/docs/cookbook/falcon-ai/eval-datasets-from-traces.mdx
+++ b/src/pages/docs/cookbook/falcon-ai/eval-datasets-from-traces.mdx
@@ -180,7 +180,7 @@ Production traces, curated and ground-truthed in one Falcon AI conversation, bec
From a single bad trace to a paste-ready prompt fix in minutes
-
+
All built-in slash commands and how to write your own
diff --git a/src/pages/docs/cookbook/langchain-langgraph.mdx b/src/pages/docs/cookbook/langchain-langgraph.mdx
index 52f1c4dc..2fc52605 100644
--- a/src/pages/docs/cookbook/langchain-langgraph.mdx
+++ b/src/pages/docs/cookbook/langchain-langgraph.mdx
@@ -116,7 +116,7 @@ Instrumentation of such project requires 3 steps:
- **`mapping`**: This dictionary maps the required inputs for the evaluation to specific attributes of the operation.
- **`custom_eval_name`**: A user-defined name for the specific evaluation instance.
- > Click [**here**](https://docs.futureagi.com/docs/prototype/evals) to learn more about the evals provided by Future AGI
+ > Click [**here**](/docs/evaluation/builtin) to learn more about the evals provided by Future AGI
>
2. **Setting Up Trace Provider:**
diff --git a/src/pages/docs/cookbook/mcp/debug-traces-from-ide.mdx b/src/pages/docs/cookbook/mcp/debug-traces-from-ide.mdx
index 3e07374e..1c89fe9e 100644
--- a/src/pages/docs/cookbook/mcp/debug-traces-from-ide.mdx
+++ b/src/pages/docs/cookbook/mcp/debug-traces-from-ide.mdx
@@ -52,7 +52,7 @@ Add to `~/.cursor/mcp.json`:
}
```
-Or use the [one-click install link](/docs/quickstart/setup-mcp-server) on the setup page.
+Or use the [one-click install link](/docs/falcon-ai/guides/use-the-mcp-server) on the setup page.
@@ -180,7 +180,7 @@ You connected Future AGI's MCP server to your IDE, asked natural-language questi
## Explore further
-
+
Full setup reference, OAuth scopes, and supported tool groups
diff --git a/src/pages/docs/cookbook/quickstart/dataset-annotation.mdx b/src/pages/docs/cookbook/quickstart/dataset-annotation.mdx
index d8361104..83e73baf 100644
--- a/src/pages/docs/cookbook/quickstart/dataset-annotation.mdx
+++ b/src/pages/docs/cookbook/quickstart/dataset-annotation.mdx
@@ -210,7 +210,7 @@ You can now create annotation views, define labels, assign annotators, and log a
## Next steps
-
+
Full annotation reference
diff --git a/src/pages/docs/cookbook/quickstart/dynamic-dataset-columns.mdx b/src/pages/docs/cookbook/quickstart/dynamic-dataset-columns.mdx
index 6f02a5d8..f774249a 100644
--- a/src/pages/docs/cookbook/quickstart/dynamic-dataset-columns.mdx
+++ b/src/pages/docs/cookbook/quickstart/dynamic-dataset-columns.mdx
@@ -178,7 +178,7 @@ You can add `elif` branches between `if` and `else` for more granular routing; e
-For all six dynamic column types in detail, see [Create Dynamic Column](/docs/dataset/concept/dynamic-column).
+For all six dynamic column types in detail, see [Create Dynamic Column](/docs/dataset/concepts/static-and-dynamic-columns).
@@ -208,7 +208,7 @@ You can now enrich any dataset with AI-generated columns, vector-retrieved conte
A/B test prompts
-
+
Full column type reference
diff --git a/src/pages/docs/cookbook/quickstart/prompt-optimization.mdx b/src/pages/docs/cookbook/quickstart/prompt-optimization.mdx
index 2be00991..2e1cb5d8 100644
--- a/src/pages/docs/cookbook/quickstart/prompt-optimization.mdx
+++ b/src/pages/docs/cookbook/quickstart/prompt-optimization.mdx
@@ -286,12 +286,12 @@ This guide uses **MetaPrompt** and **Bayesian Search**, but FutureAGI offers six
| Optimizer | Best for | How it works |
|---|---|---|
-| [**Meta-Prompt**](/docs/optimization/optimizers/meta-prompt) | General prompt improvement | A teacher LLM iteratively rewrites the prompt based on eval feedback |
-| [**Bayesian Search**](/docs/optimization/optimizers/bayesian-search) | Few-shot example selection | Uses Bayesian optimization to find the best number and combination of examples |
-| [**ProTeGi**](/docs/optimization/optimizers/protegi) | Targeted prompt editing | Generates localized edits to specific parts of the prompt, then tests each |
-| [**GEPA**](/docs/optimization/optimizers/gepa) | Exploring diverse prompt styles | Evolutionary approach — breeds, mutates, and selects prompts over generations |
-| [**PromptWizard**](/docs/optimization/optimizers/promptwizard) | Multi-stage refinement | Combines critique, refinement, and example synthesis in a structured pipeline |
-| [**Random Search**](/docs/optimization/optimizers/random-search) | Quick baseline comparison | Generates random prompt variants and picks the best — useful as a sanity check |
+| [**Meta-Prompt**](/docs/optimization/reference/optimizers/meta-prompt) | General prompt improvement | A teacher LLM iteratively rewrites the prompt based on eval feedback |
+| [**Bayesian Search**](/docs/optimization/reference/optimizers/bayesian-search) | Few-shot example selection | Uses Bayesian optimization to find the best number and combination of examples |
+| [**ProTeGi**](/docs/optimization/reference/optimizers/protegi) | Targeted prompt editing | Generates localized edits to specific parts of the prompt, then tests each |
+| [**GEPA**](/docs/optimization/reference/optimizers/gepa) | Exploring diverse prompt styles | Evolutionary approach — breeds, mutates, and selects prompts over generations |
+| [**PromptWizard**](/docs/optimization/reference/optimizers/promptwizard) | Multi-stage refinement | Combines critique, refinement, and example synthesis in a structured pipeline |
+| [**Random Search**](/docs/optimization/reference/optimizers/random-search) | Quick baseline comparison | Generates random prompt variants and picks the best — useful as a sanity check |
Not sure which to pick? Start with **Meta-Prompt** for instruction tuning or **Bayesian Search** for few-shot tasks. See the [Optimizers Overview](/docs/optimization) for a detailed comparison and decision tree.
@@ -319,7 +319,7 @@ You can now automatically optimize any prompt using MetaPromptOptimizer or Bayes
Version and serve prompts
-
+
Run optimization from the Future AGI UI
diff --git a/src/pages/docs/cookbook/quickstart/protect-guardrails.mdx b/src/pages/docs/cookbook/quickstart/protect-guardrails.mdx
index 47c49661..6a70b746 100644
--- a/src/pages/docs/cookbook/quickstart/protect-guardrails.mdx
+++ b/src/pages/docs/cookbook/quickstart/protect-guardrails.mdx
@@ -74,7 +74,7 @@ print(result["messages"]) # "What are your business hours?"
```
-`failed_rule` and `reasons` are always **lists** — even when only one rule triggers. For full details on all return keys, see [Protect API Reference](/docs/protect/concepts/concept).
+`failed_rule` and `reasons` are always **lists** — even when only one rule triggers. For full details on all return keys, see [Protect API Reference](/docs/protect/concepts/understanding-protect).
@@ -137,7 +137,7 @@ print(result["failed_rule"]) # ["security", "data_privacy_compliance"]
print(result["reasons"][0]) # "Detected instruction override attempt..."
```
-The four available metrics are `content_moderation`, `security`, `data_privacy_compliance`, and `bias_detection`. See [Protect How-To](/docs/protect/features/run-protect) for what each metric catches.
+The four available metrics are `content_moderation`, `security`, `data_privacy_compliance`, and `bias_detection`. See [Protect How-To](/docs/protect/guides/run-protect-from-the-sdk) for what each metric catches.
@@ -232,7 +232,7 @@ print(result["status"]) # "passed"
```
-Use standard Protect for accuracy-critical flows (user-facing chatbots, compliance). Use Protect Flash for high-volume pipelines (batch screening, log analysis). See [Protect vs Protect Flash](/docs/protect/concepts/concept) for a detailed comparison.
+Use standard Protect for accuracy-critical flows (user-facing chatbots, compliance). Use Protect Flash for high-volume pipelines (batch screening, log analysis). See [Protect vs Protect Flash](/docs/protect/concepts/understanding-protect) for a detailed comparison.
@@ -253,10 +253,10 @@ You can now screen user inputs and AI outputs for prompt injection, PII, toxicit
## Next steps
-
+
All safety metrics
-
+
How Protect works
diff --git a/src/pages/docs/cookbook/quickstart/prototype-llm-app.mdx b/src/pages/docs/cookbook/quickstart/prototype-llm-app.mdx
deleted file mode 100644
index 9b29ffa5..00000000
--- a/src/pages/docs/cookbook/quickstart/prototype-llm-app.mdx
+++ /dev/null
@@ -1,316 +0,0 @@
----
-title: "Prototype and Iterate on LLM Applications"
-description: "Register a Prototype project with automatic span evaluation, iterate with versioned prompts, and compare versions before deploying to production."
----
-
-
-Register a Prototype project with automatic span evaluation, iterate with versioned prompts, compare versions side by side, and choose a winner before deploying to production.
-
-
-
-
-
-
-
-| Time | Difficulty | Package |
-|------|-----------|---------|
-| 15 min | Intermediate | `fi-instrumentation-otel` |
-
-
-- FutureAGI account → [app.futureagi.com](https://app.futureagi.com)
-- API keys: `FI_API_KEY` and `FI_SECRET_KEY` (see [Get your API keys](/docs/admin-settings))
-- Python 3.9+
-- OpenAI API key
-
-
-## Install
-
-```bash
-pip install fi-instrumentation-otel traceAI-openai openai
-```
-
-```bash
-export FI_API_KEY="your-api-key"
-export FI_SECRET_KEY="your-secret-key"
-export OPENAI_API_KEY="your-openai-api-key"
-```
-
----
-
-## What is Prototype?
-
-Prototype lets you test different LLM configurations, prompts, and parameters in a controlled environment before deploying to production. Each run is a **version**: you compare versions side by side on evaluation scores, cost, and latency, then choose a winner.
-
-## Tutorial
-
-
-
-
-`register()` creates a tracer provider connected to FutureAGI. Setting `project_type=ProjectType.EXPERIMENT` creates a Prototype project. The `project_version_name` tags all traces from this run as a distinct version you can compare later.
-
-`EvalTag` objects define which evaluations run automatically on every matching span, with no manual eval calls needed.
-
-```python
-from fi_instrumentation import register
-from fi_instrumentation.fi_types import (
- ProjectType,
- EvalName,
- EvalTag,
- EvalTagType,
- EvalSpanKind,
- ModelChoices,
-)
-
-trace_provider = register(
- project_type=ProjectType.EXPERIMENT,
- project_name="support-bot-prototype",
- project_version_name="v1-baseline",
- eval_tags=[
- EvalTag(
- eval_name=EvalName.COMPLETENESS,
- type=EvalTagType.OBSERVATION_SPAN,
- value=EvalSpanKind.LLM,
- model=ModelChoices.TURING_FLASH,
- custom_eval_name="completeness_check",
- mapping={
- "input": "llm.input_messages.1.message.content",
- "output": "llm.output_messages.0.message.content",
- },
- ),
- EvalTag(
- eval_name=EvalName.SUMMARY_QUALITY,
- type=EvalTagType.OBSERVATION_SPAN,
- value=EvalSpanKind.LLM,
- model=ModelChoices.TURING_FLASH,
- custom_eval_name="response_quality",
- mapping={
- "input": "llm.input_messages.1.message.content",
- "output": "llm.output_messages.0.message.content",
- },
- ),
- ],
-)
-```
-
-Each `EvalTag` has:
-- `eval_name`: the built-in evaluation to run (e.g. `EvalName.COMPLETENESS`, `EvalName.SUMMARY_QUALITY`)
-- `type`: where to apply the eval (`EvalTagType.OBSERVATION_SPAN`)
-- `value`: which span kind to evaluate (`EvalSpanKind.LLM`)
-- `mapping`: maps eval input keys to span attribute paths
-- `model`: the FutureAGI eval model to use
-- `custom_eval_name`: a label for this eval tag (must be unique per project)
-
-
-
-
-Patch the OpenAI client with `OpenAIInstrumentor` so every API call is automatically traced and evaluated against your `EvalTag` configuration.
-
-```python
-from traceai_openai import OpenAIInstrumentor
-from openai import OpenAI
-
-OpenAIInstrumentor().instrument(tracer_provider=trace_provider)
-
-client = OpenAI()
-
-questions = [
- "How do I reset my password?",
- "What is your refund policy?",
- "Can I upgrade my plan mid-cycle?",
-]
-
-for q in questions:
- response = client.chat.completions.create(
- model="gpt-4o-mini",
- messages=[
- {"role": "system", "content": "You are a helpful customer support agent. Answer concisely."},
- {"role": "user", "content": q},
- ],
- )
- print(f"Q: {q}")
- print(f"A: {response.choices[0].message.content}\n")
-
-trace_provider.force_flush()
-```
-
-Expected output:
-```
-Q: How do I reset my password?
-A: Go to the login page, click "Forgot Password," enter your email, and follow the reset link sent to your inbox.
-
-Q: What is your refund policy?
-A: We offer full refunds within 30 days of purchase. After 30 days, refunds are prorated.
-
-Q: Can I upgrade my plan mid-cycle?
-A: Yes, you can upgrade anytime. The price difference is prorated for the remainder of your billing cycle.
-```
-
-
-
-
-Go to [app.futureagi.com](https://app.futureagi.com), select **Prototype** (left sidebar under BUILD), and click your project **support-bot-prototype** to see version **v1-baseline**.
-
-The dashboard shows:
-- Every traced span with its input, output, token count, and latency
-- Evaluation scores from your `EvalTag` configuration (`completeness_check` and `response_quality`) displayed alongside each span
-
-
-
-
-
-
-This is where rapid iteration happens. Register a new version with a different `project_version_name` and run the same queries with an improved prompt. Each version is a separate experiment you can compare.
-
-
-Each call to `register()` creates a new tracer provider. Run Version 2 in a separate script or after the Version 1 script completes — do not call `register()` twice in the same process.
-
-
-```python
-from fi_instrumentation import register
-from fi_instrumentation.fi_types import (
- ProjectType,
- EvalName,
- EvalTag,
- EvalTagType,
- EvalSpanKind,
- ModelChoices,
-)
-from traceai_openai import OpenAIInstrumentor
-from openai import OpenAI
-
-trace_provider_v2 = register(
- project_type=ProjectType.EXPERIMENT,
- project_name="support-bot-prototype",
- project_version_name="v2-detailed",
- eval_tags=[
- EvalTag(
- eval_name=EvalName.COMPLETENESS,
- type=EvalTagType.OBSERVATION_SPAN,
- value=EvalSpanKind.LLM,
- model=ModelChoices.TURING_FLASH,
- custom_eval_name="completeness_check",
- mapping={
- "input": "llm.input_messages.1.message.content",
- "output": "llm.output_messages.0.message.content",
- },
- ),
- EvalTag(
- eval_name=EvalName.SUMMARY_QUALITY,
- type=EvalTagType.OBSERVATION_SPAN,
- value=EvalSpanKind.LLM,
- model=ModelChoices.TURING_FLASH,
- custom_eval_name="response_quality",
- mapping={
- "input": "llm.input_messages.1.message.content",
- "output": "llm.output_messages.0.message.content",
- },
- ),
- ],
-)
-
-OpenAIInstrumentor().uninstrument()
-OpenAIInstrumentor().instrument(tracer_provider=trace_provider_v2)
-
-client = OpenAI()
-
-questions = [
- "How do I reset my password?",
- "What is your refund policy?",
- "Can I upgrade my plan mid-cycle?",
-]
-
-for q in questions:
- response = client.chat.completions.create(
- model="gpt-4o-mini",
- messages=[
- {
- "role": "system",
- "content": (
- "You are a knowledgeable customer support agent. "
- "Provide detailed, step-by-step answers. "
- "Include any relevant edge cases or exceptions. "
- "End with a follow-up question to confirm the issue is resolved."
- ),
- },
- {"role": "user", "content": q},
- ],
- )
- print(f"Q: {q}")
- print(f"A: {response.choices[0].message.content}\n")
-
-trace_provider_v2.force_flush()
-```
-
-Expected output:
-```
-Q: How do I reset my password?
-A: Here's how to reset your password step by step:
-1. Go to our login page at app.example.com
-2. Click "Forgot Password" below the sign-in button
-3. Enter the email address associated with your account
-4. Check your inbox for a reset link (check spam if you don't see it within 5 minutes)
-5. Click the link and enter your new password
-
-Note: The reset link expires after 24 hours. If it expires, repeat the process.
-
-Is there anything else about your account access I can help with?
-
-Q: What is your refund policy?
-...
-```
-
-
-
-
-Back in the Prototype dashboard, your project now shows two versions: **v1-baseline** and **v2-detailed**.
-
-Click any version to see its individual traces and eval scores. The project overview shows aggregate metrics across all versions — average eval scores, latency, token usage, and cost — so you can compare at a glance.
-
-
-
-
-
-
-Once you have compared evaluation scores, latency, and cost across versions, choose a winner.
-
-1. Go to **Prototype** → click your project
-2. Click **Choose Winner** — a **Winner Settings** drawer opens
-3. Under **Evaluation Metrics**, adjust the importance slider (0 = Not Important, 10 = Very Important) for each eval — `completeness_check` and `response_quality`
-4. Under **System Metrics**, adjust the importance sliders for **Avg Cost** and **Avg Latency**
-5. Click **Choose Winner** to rank all versions
-
-The version with the highest weighted score across your chosen importance values is selected as the winner.
-
-{/* The recording above (Step 5) also covers the Choose Winner flow. */}
-
-
-
-
-## What you built
-
-
-You can now register a Prototype project, auto-evaluate spans with EvalTags, iterate with versioned prompts, compare versions, and choose the best one for production.
-
-
-- Registered a Prototype project with `ProjectType.EXPERIMENT` and automatic span evaluation via `EvalTag`
-- Ran a baseline OpenAI app (v1) and saw completeness and response quality scores in the dashboard
-- Iterated with a new prompt version (v2) using a different `project_version_name`
-- Compared both versions on eval scores, latency, and cost in the Prototype dashboard
-- Chose the winning version using weighted metric comparison
-
-## Next steps
-
-
-
- Docs and version management
-
-
- EvalTag configurations
-
-
- UI-first prompt comparison
-
-
- Custom spans and metadata
-
-
diff --git a/src/pages/docs/cookbook/text-to-sql.mdx b/src/pages/docs/cookbook/text-to-sql.mdx
index 17ffe984..4937a979 100644
--- a/src/pages/docs/cookbook/text-to-sql.mdx
+++ b/src/pages/docs/cookbook/text-to-sql.mdx
@@ -375,7 +375,7 @@ def setup_database():
- `DETECT_HALLUCINATION`: Identifies instances where the agent generates SQL that references non-existent tables, columns, or relationships that aren't present in the database schema.
- `table_checker`: A custom evaluation that verifies whether the SQL queries reference the appropriate tables needed to satisfy the user's request, ensuring optimal join patterns and table selection.
- > **Click [here](https://docs.futureagi.com/docs/prototype/evals) to learn more about the evals provided by Future AGI**
+ > **Click [here](/docs/evaluation/builtin) to learn more about the evals provided by Future AGI**
>
- The **`eval_tags`** list contains multiple instances of **`EvalTag`**. Each **`EvalTag`** represents a specific evaluation configuration to be applied during runtime, encapsulating all necessary parameters for the evaluation process.
- Parameters of **`EvalTag`** :
@@ -385,15 +385,15 @@ def setup_database():
- **`EvalSpanKind.TOOL`**: For operations involving tools.
- **`eval_name`**: The name of the evaluation to be performed.
- > Click [**here**](https://docs.futureagi.com/docs/prototype/evals) to get complete list of evals provided by Future AGI
+ > Click [**here**](/docs/evaluation/builtin) to get complete list of evals provided by Future AGI
>
- **`config`**: Dictionary for providing specific configurations for the evaluation. An empty dictionary means that default configuration parameters will be used.
- Click [**here**](https://docs.futureagi.com/docs/prototype/evals) to learn more about what config is required for corresponding evals
+ Click [**here**](/docs/evaluation/builtin) to learn more about what config is required for corresponding evals
- **`mapping`**: This dictionary maps the required inputs for the evaluation to specific attributes of the operation.
- Click [**here**](https://docs.futureagi.com/docs/prototype/evals) to learn more about what inputs are required for corresponding evals
+ Click [**here**](/docs/evaluation/builtin) to learn more about what inputs are required for corresponding evals
- **`custom_eval_name`**: A user-defined name for the specific evaluation instance.
- `model`: LLM model name required to perform the evaluation. Such as `TURING_LARGE`, which is a proprietary model provided by Future AGI.
diff --git a/src/pages/docs/dataset/concept/dynamic-column.mdx b/src/pages/docs/dataset/concept/dynamic-column.mdx
deleted file mode 100644
index 8a53ef9d..00000000
--- a/src/pages/docs/dataset/concept/dynamic-column.mdx
+++ /dev/null
@@ -1,62 +0,0 @@
----
-title: "Dynamic Columns: Auto-Generated Dataset Values in Future AGI"
-description: "Dataset columns auto-generated by running LLM prompts, evaluations, vector retrieval, entity extraction, or custom Python code against every row."
----
-
-## About
-
-A dynamic column is generated automatically by the platform. Instead of entering data yourself, you configure a method (like running an LLM prompt or an evaluation) and the platform computes a value for every row.
-
-For example, starting with two [static columns](/docs/dataset/concept/static-column):
-
-| user_query | expected_answer | model_response | is_correct |
-|---|---|---|---|
-| What is the capital of France? | Paris | Paris | true |
-| Who wrote Hamlet? | Shakespeare | William Shakespeare | true |
-
-Here `model_response` is a dynamic column created by running a prompt against each `user_query`. And `is_correct` is another dynamic column created by running an evaluation that compares `model_response` to `expected_answer`.
-
-Dynamic columns can be regenerated at any time. If you change the prompt or switch models, you can re-run the column and the values update across all rows.
-
----
-
-## When to use
-
-- **Get model outputs**: Run an LLM on every row and store the responses for comparison or evaluation
-- **Score outputs**: Run evaluations and store the results (pass/fail, scores, explanations) alongside your data
-- **Extract structured data**: Pull entities, JSON keys, or classifications out of unstructured text columns
-- **Enrich with external data**: Call APIs or vector databases to add context to each row
-- **Transform data**: Apply custom Python logic to compute derived values
-
----
-
-## Supported Methods
-
-| Method | What it does |
-|---|---|
-| Run Prompt | Run an LLM prompt that can reference other columns as variables. [Learn more](/docs/dataset/features/run-prompt) |
-| Vector Retrieval | Connect to a vector database and retrieve the top-k chunks for a query |
-| Entity Extraction | Extract named entities (people, organizations, locations) from text columns using a model |
-| JSON Key Extraction | Parse a JSON column and extract specific keys or nested values |
-| Custom Code Execution | Write and run Python code for transformations or complex operations |
-| Text Classification | Assign categories or labels to text using a model |
-| API Calls | Call an external API endpoint for every row and store the response |
-| Conditional Logic | Apply different actions based on conditions (if/else branching across rows) |
-
----
-
-## How It Works
-
-1. Choose a dynamic column method from the list above
-2. Configure the method (select a model, write a prompt, define the logic)
-3. Map input columns (e.g. use `user_query` as the input to your prompt)
-4. Run the column. The platform processes all rows in parallel and fills in the values.
-5. View the results in your dataset. Re-run anytime to refresh.
-
----
-
-## Next Steps
-
-- [Static Columns](/docs/dataset/concept/static-column): Columns with fixed data you provide directly
-- [Run Prompt in Dataset](/docs/dataset/features/run-prompt): The most common dynamic column method
-- [Experiments](/docs/dataset/features/experiments): Compare dynamic column results across different configurations
\ No newline at end of file
diff --git a/src/pages/docs/dataset/concept/static-column.mdx b/src/pages/docs/dataset/concept/static-column.mdx
deleted file mode 100644
index e5cb4014..00000000
--- a/src/pages/docs/dataset/concept/static-column.mdx
+++ /dev/null
@@ -1,56 +0,0 @@
----
-title: "Static Columns: Fixed Dataset Values in Future AGI"
-description: "Dataset columns for storing fixed test inputs, expected outputs, labels, and metadata. Supports 9 data types including text, JSON, image, and audio."
----
-
-## About
-
-A static column holds data that you provide directly. This includes inputs, expected outputs, labels, categories, or any fixed values. Unlike [dynamic columns](/docs/dataset/concept/dynamic-column), static columns don't run any computation. They only change when you update them manually or through the SDK.
-
-For example, in this dataset the first three columns are static:
-
-| user_query | expected_answer | category | model_response |
-|---|---|---|---|
-| What is the capital of France? | Paris | geography | *(dynamic)* |
-| Summarize this article | A concise summary of... | summarization | *(dynamic)* |
-
-You add `user_query`, `expected_answer`, and `category` yourself. The `model_response` column would be a [dynamic column](/docs/dataset/concept/dynamic-column) generated by running a prompt.
-
----
-
-## When to use
-
-- **Test inputs and expected outputs**: Store the queries and ground truth answers for evaluation
-- **Labels and categories**: Tag rows with classifications (e.g. "easy", "hard", "geography", "math")
-- **Default values**: Pre-fill rows with consistent starting data when setting up a dataset
-- **Metadata**: Store context like source, timestamp, or user ID alongside your test data
-
----
-
-## Supported Data Types
-
-| Type | Description |
-|---|---|
-| `text` | Strings and free-form text |
-| `integer` | Whole numbers |
-| `float` | Decimal numbers |
-| `boolean` | True or false |
-| `array` | Lists of values |
-| `json` | Structured JSON objects |
-| `image` | Image file references |
-| `audio` | Audio file references |
-| `datetime` | Date and time values |
-
----
-
-## How to Add a Static Column
-
-You can add static columns through the UI or when creating a dataset via the SDK. See [Add Columns to Dataset](/docs/dataset/features/add-columns) for step-by-step instructions.
-
----
-
-## Next Steps
-
-- [Dynamic Columns](/docs/dataset/concept/dynamic-column): Columns generated by prompts, evaluations, or models
-- [Add Columns](/docs/dataset/features/add-columns): Add new columns to an existing dataset
-- [Create a Dataset](/docs/dataset/features/create): Start a new dataset from scratch
\ No newline at end of file
diff --git a/src/pages/docs/dataset/concept/synthetic-data.mdx b/src/pages/docs/dataset/concept/synthetic-data.mdx
deleted file mode 100644
index 5f28844e..00000000
--- a/src/pages/docs/dataset/concept/synthetic-data.mdx
+++ /dev/null
@@ -1,63 +0,0 @@
----
-title: "Synthetic Data Generation for AI Testing in Future AGI"
-description: "Generate schema-driven test datasets without using real user data. Define column types, constraints, and descriptions, then generate rows using Future AGI."
----
-
-## About
-
-Synthetic data is artificially generated data that follows real-world patterns without using actual user data. In Future AGI, you define a schema (columns, types, descriptions, and constraints) and the platform generates rows that match your specification.
-
-For example, defining this schema:
-
-| Column | Type | Description |
-|---|---|---|
-| customer_query | text | A realistic customer support question |
-| sentiment | text | One of: positive, negative, neutral |
-| priority | integer | 1 (low) to 5 (urgent) |
-
-Produces rows like:
-
-| customer_query | sentiment | priority |
-|---|---|---|
-| I haven’t received my order and it’s been two weeks | negative | 4 |
-| Can I change the shipping address on my recent order? | neutral | 2 |
-| Your product is fantastic, just wanted to say thanks! | positive | 1 |
-
-The generated data follows the constraints you set (sentiment is always one of three values, priority stays in range) while producing varied, realistic content.
-
----
-
-## When to use
-
-- **No real data available**: You’re building a new feature and don’t have production data yet
-- **Privacy constraints**: Real data contains PII or sensitive information that can’t be used for testing
-- **Edge case testing**: You need specific scenarios (angry customers, rare errors, multilingual queries) that are hard to find in real data
-- **Scale testing**: You need thousands of rows to stress-test evaluations or prompts
-- **Balanced datasets**: Real data is skewed (e.g. 95% positive reviews) and you need more balanced distributions
-
----
-
-## How It Works
-
-1. Define the schema: column names, data types, and descriptions
-2. Set constraints: value ranges, categorical options, patterns
-3. Optionally connect a [Knowledge Base](/docs/knowledge-base) to ground generation with your own documents
-4. Choose the number of rows to generate
-5. The platform generates the dataset. You can review, edit, and use it immediately.
-
----
-
-## Key Properties
-
-- **Schema-driven**: You control the structure. Every column has a type, description, and optional constraints that guide generation.
-- **Realistic distribution**: Generated data follows natural patterns and distributions, not random values. Descriptions give the generator context to produce relevant content.
-- **Safe by default**: Generated data does not contain real PII, credentials, or sensitive information.
-
----
-
-## Next Steps
-
-- [Generate Synthetic Data](/docs/quickstart/generate-synthetic-data): Step-by-step quickstart for creating your first synthetic dataset
-- [Static Columns](/docs/dataset/concept/static-column): How static columns store the data you provide
-- [Dynamic Columns](/docs/dataset/concept/dynamic-column): How to add model outputs and evaluations on top of your synthetic data
-- [Knowledge Base](/docs/knowledge-base): Ground synthetic generation with your own documents
\ No newline at end of file
diff --git a/src/pages/docs/dataset/concept/understanding-dataset.mdx b/src/pages/docs/dataset/concept/understanding-dataset.mdx
deleted file mode 100644
index 056c58ab..00000000
--- a/src/pages/docs/dataset/concept/understanding-dataset.mdx
+++ /dev/null
@@ -1,77 +0,0 @@
----
-title: "Future AGI Datasets: Structure, Column Types, and Lifecycle"
-description: "Each row is one example; each column is an attribute. Datasets are the foundation for running prompts, evals, experiments, and optimizations in Future AGI."
----
-
-## About
-
-A dataset in Future AGI is a table of structured data. Each row is one example (e.g. a user query and its expected answer). Each column is an attribute (e.g. "input", "expected_output", "model_response", "score"). Datasets are the foundation for running prompts, evaluations, experiments, and optimizations.
-
-Here's what a simple dataset looks like:
-
-| input | expected_output | model_response | is_correct |
-|---|---|---|---|
-| What is the capital of France? | Paris | Paris | true |
-| Who wrote Hamlet? | Shakespeare | William Shakespeare | true |
-| What is 2+2? | 4 | The answer is 4 | true |
-
-The first two columns (input, expected_output) are [static columns](/docs/dataset/concept/static-column) that you add manually. The last two (model_response, is_correct) are [dynamic columns](/docs/dataset/concept/dynamic-column) generated by running a prompt and an evaluation against each row.
-
----
-
-## Structure
-
-Every dataset has three core components:
-
-- **Rows**: Each row is one data point or test case. You can add rows manually, import from files, generate them synthetically, or pull them from production traces.
-- **Columns**: Each column defines an attribute. Columns have a name, a data type (text, number, boolean, JSON, etc.), and are either static (you provide the data) or dynamic (the platform generates it).
-- **Metadata**: Each dataset has a name, description, and organization-level permissions that control who can view and edit it.
-
----
-
-## How to Create a Dataset
-
-There are several ways to get data into a dataset:
-
-- **Manual creation**: Define the structure and add rows through the UI or SDK. [Learn more](/docs/dataset/features/create)
-- **File import**: Upload CSV, Excel, JSON, or JSONL files. [Learn more](/docs/dataset/features/create)
-- **Synthetic generation**: Describe the schema and let the platform generate realistic test data. [Learn more](/docs/dataset/concept/synthetic-data)
-- **From HuggingFace**: Import existing datasets from HuggingFace directly. [Learn more](/docs/cookbook/quickstart/huggingface-dataset-import)
-- **From production traces**: Convert observed production data from the Observe module into datasets for regression testing. [Learn more](/docs/observe)
-
----
-
-## Dataset Lifecycle
-
-### 1. Create
-
-Start with a schema (columns and types) and populate it with data using any of the methods above.
-
-### 2. Enrich
-
-Add more columns to your dataset over time:
-
-- **Run prompts**: Send each row through an LLM and store the responses as a new column. [Learn more](/docs/dataset/features/run-prompt)
-- **Run evaluations**: Score model outputs using 70+ built-in metrics. Results are stored as new columns. [Learn more](/docs/evaluation)
-- **Add annotations**: Manually label rows with custom tags and scores. Future AGI also supports auto-annotations that learn from your labels. [Learn more](/docs/dataset/features/annotate)
-
-### 3. Experiment
-
-Use the same dataset to compare different prompts, models, or configurations side by side. Each experiment run adds new columns so you can see results next to each other. [Learn more](/docs/dataset/features/experiments)
-
-### 4. Maintain
-
-Datasets evolve over time. You can:
-
-- Add or remove columns without disrupting existing data
-- Add new rows as you discover edge cases
-- Archive or delete old datasets to keep your workspace clean
-
----
-
-## Next Steps
-
-- [Static Columns](/docs/dataset/concept/static-column): Data you add directly to your dataset
-- [Dynamic Columns](/docs/dataset/concept/dynamic-column): Data generated by prompts, evaluations, or models
-- [Synthetic Data](/docs/dataset/concept/synthetic-data): Generate realistic test data from a schema
-- [Create a Dataset](/docs/dataset/features/create): Get started with your first dataset
\ No newline at end of file
diff --git a/src/pages/docs/dataset/concepts/static-and-dynamic-columns.mdx b/src/pages/docs/dataset/concepts/static-and-dynamic-columns.mdx
new file mode 100644
index 00000000..848ba38c
--- /dev/null
+++ b/src/pages/docs/dataset/concepts/static-and-dynamic-columns.mdx
@@ -0,0 +1,62 @@
+---
+title: "Static & Dynamic Columns"
+description: "Whether a column holds values you set yourself, or values a producer computes for you"
+---
+
+## Where a column's values come from
+
+Every [column](/docs/dataset/concepts/understanding-datasets) is either **static** or **dynamic**, and that's a property of the column itself, not of any single row inside it. The difference comes down to where its values come from.
+
+A static column holds values you supply. You type them in, paste them, or set them through the SDK, and a cell only changes when you go back and edit it.
+
+A dynamic column doesn't hold values you typed, it holds something that fills them for you: [a prompt you run over every row](/docs/dataset/guides/run-a-prompt-on-every-row), an evaluation, an API call, and more. Every cell in the column is whatever that produced for that row, not something you set by hand. The full set of things a dynamic column can run is cataloged in [Dynamic column methods](/docs/dataset/reference/dynamic-column-methods).
+
+Take a dataset with four columns:
+
+| user_query | expected_answer | model_response | is_correct |
+|---|---|---|---|
+| What is the capital of France? | Paris | Paris | true |
+| Who wrote Hamlet? | Shakespeare | William Shakespeare | true |
+
+`user_query` and `expected_answer` are static, you wrote them in. `model_response` is dynamic: a prompt behind the column answers `user_query` for every row. `is_correct` is dynamic too: an evaluation behind it compares `model_response` against `expected_answer`.
+
+## Mental model: producer or no producer
+
+ ST["Static you fill it"]
+ COL --> DY["Dynamic something fills it"]
+ ST -->|"you set it"| CS1["Cell · row 1"]
+ ST -->|"you set it"| CS2["Cell · row 2"]
+ DY -->|"runs"| PR["A prompt, or an evaluation"]
+ PR -->|"fills"| CD1["Cell · row 1"]
+ PR -->|"fills"| CD2["Cell · row 2"]`} />
+
+## What follows from having a producer behind the column
+
+Several consequences fall directly out of that difference.
+
+**It carries a status while the producer runs.** A static column has no run to track, so it has no status to show. A dynamic column does: while its producer is working, the column sits in a running state, and if the producer fails, the column shows failed. That status is the tell for whether you're looking at a value you can trust yet.
+
+**It can be re-run, and every row changes at once.** The producer behind a dynamic column doesn't disappear after the first run. Change the prompt, switch the model, fix the eval config, then re-run the column, and every cell it owns recomputes together. Editing a static column, by contrast, is you overwriting one cell at a time; nothing else moves.
+
+**Deleting it takes the producer with it.** A dynamic column isn't just the column, it's the column plus the producer generating it. Delete the column and its producer goes too, along with anything else that was derived from it. Deleting a static column removes only the column and the values sitting in it, there's no producer behind it to clean up.
+
+## Why it matters
+
+Before you touch a column, it's worth knowing which kind you're looking at. The consequences above all come from the same root: touch a dynamic column and you're really touching the prompt or evaluation behind it, not just the cell or the column in front of you.
+
+## Keep exploring
+
+
+
+ Create a static or dynamic column in a dataset
+
+
+ The most common dynamic column, walked end to end
+
+
+ Methods you can point a dynamic column at
+
+
diff --git a/src/pages/docs/dataset/concepts/synthetic-data.mdx b/src/pages/docs/dataset/concepts/synthetic-data.mdx
new file mode 100644
index 00000000..294c564a
--- /dev/null
+++ b/src/pages/docs/dataset/concepts/synthetic-data.mdx
@@ -0,0 +1,102 @@
+---
+title: "Synthetic Data"
+description: "Turning a column schema into realistic dataset rows, without real production data"
+---
+
+## What synthetic data is
+
+**Synthetic data** is a [dataset's](/docs/dataset/concepts/understanding-datasets) rows generated from a schema you define, instead of rows you upload or bring in yourself. You describe the [columns](/docs/dataset/concepts/static-and-dynamic-columns) you want, their names, types, and constraints, and Future AGI generates rows that match.
+
+Define this schema for a customer-support dataset:
+
+| Column | Type | Constraints |
+|---|---|---|
+| customer_query | text | Value: a realistic customer support question |
+| sentiment | text | Categorical values: positive, negative, neutral |
+| priority | integer | Value: 1 (low) to 5 (urgent) |
+
+Generation produces rows like:
+
+| customer_query | sentiment | priority |
+|---|---|---|
+| I haven't received my order and it's been two weeks | negative | 4 |
+| Can I change the shipping address on my recent order? | neutral | 2 |
+| Your product is fantastic, just wanted to say thanks! | positive | 1 |
+
+Each column has its own Property editor for exactly this: Min Length and Max Length on most column types, Value set to Categorical for a list of allowed values, plus custom properties for anything else. Those column properties, not the column's description, are where allowed values and ranges live.
+
+
+Column properties steer the generator toward matching rows, but they aren't hard validation on the result. Skim the generated rows before you rely on them.
+
+
+To generate your first synthetic dataset hands-on, follow the [Generate synthetic data quickstart](/docs/quickstart/generate-synthetic-data).
+
+## The generation config
+
+Every synthetic dataset saves what you defined as the dataset's own generation config:
+
+- **Columns**: the schema you defined, with each column's name, type, and description
+- **Row count**: how many rows to generate
+- **Description**: what the dataset as a whole should contain
+- **Objective**: how you plan to use the dataset, so generation can match that goal
+- **Pattern**: an example or format you want the generated rows to follow
+- An optional [Knowledge Base](/docs/knowledge-base) link
+
+That's what lets you reopen a synthetic dataset later, change a column or the row count, and regenerate without rebuilding the schema from scratch. Here's how the config, the job, and the dataset's rows and state fit together:
+
+|read by| JOB
+ JOB -->|fills| ROWS
+ JOB -->|drives| STATE
+ KB -.->|grounds| JOB`} />
+
+When a Knowledge Base is connected, generation grounds rows in its content instead of relying on the schema alone.
+
+### Editing vs regenerating
+
+| Action | What happens |
+|---|---|
+| Edit the config and save | Adds the columns and rows you added, drops the columns you removed, drops rows if you lowered the row count, and leaves every other column's data as it is |
+| Regenerate | Rebuilds all rows and columns from the config (destructive, see the warning below) |
+
+
+Regenerating wipes the dataset's current rows and columns and rebuilds them fresh from the config. If you only meant to add a column or a few rows, edit and save instead.
+
+
+The saved config is what makes either possible: you're never redefining the schema by hand.
+
+## When to use synthetic data
+
+Reach for synthetic data whenever real rows are unavailable, risky to use, or lopsided for what you're testing:
+
+- **No real data yet**: you're building a new feature and don't have production rows to test against
+- **Privacy limits**: real data carries PII you can't put in a test dataset
+- **Edge cases**: you need scenarios that are rare in real traffic, like an angry customer or a multilingual query
+- **Scale**: you need thousands of rows to stress-test a prompt or eval
+- **Skewed data**: real data leans one way (mostly positive reviews) and you need a more balanced set
+
+## While it's generating, and when it fails
+
+Generation doesn't happen instantly. Because it runs as a background job, a synthetic dataset sits in a **Generating** state (or **Regenerating**, if you kicked off a rerun) with a live progress bar while the job works, and a **Configure Synthetic Data** button that reopens the schema.
+
+If the job errors out, the dataset shows a **Failed** state instead, with the same button to fix the configuration and try again. See [Dataset FAQ & fixes](/docs/dataset/troubleshooting) for what to check first.
+
+## Keep exploring
+
+
+
+ Step-by-step guide for creating a dataset, including the synthetic flow
+
+
+ How static and dynamic columns store and compute values
+
+
diff --git a/src/pages/docs/dataset/concepts/understanding-datasets.mdx b/src/pages/docs/dataset/concepts/understanding-datasets.mdx
new file mode 100644
index 00000000..29c28ac6
--- /dev/null
+++ b/src/pages/docs/dataset/concepts/understanding-datasets.mdx
@@ -0,0 +1,65 @@
+---
+title: "Understanding Datasets"
+description: "The columns, rows, and cells a dataset is built from, and who owns it"
+---
+
+## What a dataset is made of
+
+A **dataset** is what you run [prompts](/docs/dataset/guides/run-a-prompt-on-every-row), evals, and [experiments](/docs/dataset/guides/run-an-experiment) against. It owns two collections, columns and rows, plus a few fields on the dataset itself.
+
+### Columns, rows, and cells
+
+A **column** defines one attribute every row carries. Whether its values are static or dynamic, who or what fills them in, is covered in [Static & Dynamic Columns](/docs/dataset/concepts/static-and-dynamic-columns). A **row** is one example. Where a row crosses a column sits a **cell**: the stored value for that column, on that row.
+
+Take a two-column dataset with columns `input` and `model_response`. Both rows get a cell in each column: four cells total, all belonging to the same dataset. Add a third row and both columns grow a new cell; add a third column and both rows do too, which is what keeps the grid rectangular no matter how many of its columns are dynamic.
+
+ C1["Column · input"]
+ D --> C2["Column · model_response"]
+ D --> R1["Row 1"]
+ D --> R2["Row 2"]
+ C1 --> X11["Cell · Row 1 × input"]
+ R1 --> X11
+ C1 --> X21["Cell · Row 2 × input"]
+ R2 --> X21
+ C2 --> X12["Cell · Row 1 × model_response"]
+ R1 --> X12
+ C2 --> X22["Cell · Row 2 × model_response"]
+ R2 --> X22`} />
+
+### Row and column order
+
+Row order isn't insertion order. Every row carries its own row-level `order`, an explicit integer that fixes where it sits top to bottom. Column layout works the same way one level up: the dataset carries a dataset-level `column_order` array that fixes the left-to-right order of columns, and it's pruned automatically when a [column is deleted](/docs/dataset/guides/manage-datasets).
+
+### How cell values are stored
+
+A cell's value is always stored as text, whatever [the column's data type](/docs/dataset/reference/limits-and-data-types). A JSON column's value is JSON-stringified before it's stored, and a media column, an image or an audio file, holds a URL string rather than the file itself.
+
+## Ownership and scoping
+
+A dataset always belongs to exactly one organization; there's no such thing as a dataset with no owning org. A workspace is optional: a dataset can sit inside one workspace for scoping, or none at all.
+
+`model_type` fixes what kind of data the dataset is built for: generative text by default, or image, audio, video, and structured types such as classification and ranking.
+
+## Why it matters
+
+- You can resort rows for review without disturbing anything else in the dataset; `order` is separate from when a row was created or which cells it holds
+- Deleting a column cleans up its position in the layout automatically, so nothing is left pointing at a column that no longer exists
+- Anything that reads a cell back, a prompt template, an eval, an export, gets a string and has to parse or fetch it for JSON and media columns
+- Access to a dataset follows organization membership first; the optional workspace narrows that further
+
+## Keep exploring
+
+
+
+ Where a column's values come from, and what changes when it's dynamic
+
+
+ Generate realistic rows from a schema instead of writing them by hand
+
+
+ Every way to get a dataset that exists and has data in it
+
+
diff --git a/src/pages/docs/dataset/features/add-columns.mdx b/src/pages/docs/dataset/features/add-columns.mdx
deleted file mode 100644
index 759314b6..00000000
--- a/src/pages/docs/dataset/features/add-columns.mdx
+++ /dev/null
@@ -1,188 +0,0 @@
----
-title: "Adding Static and Dynamic Columns to a Dataset"
-description: "Add static columns for fixed values or dynamic columns whose values are computed from other columns or external operations."
----
-
-## About
-
-Adding a column extends your dataset with a new field. Columns can be of two kinds:
-
-- **[Static columns](/docs/dataset/concept/static-column)**: Store fixed values (text, numbers, boolean, array, JSON) that you enter or paste. They do not require computation; you edit cells manually.
-- **[Dynamic columns](/docs/dataset/concept/dynamic-column)**: Values are computed or fetched when you need them (e.g. from an LLM prompt, vector DB, API, custom code, or from existing columns). You configure the type, test, then create; the system fills the column row by row.
-
-Both are added via **+ Add Columns** in your dataset.
-
-## When to use
-
-- **Store reference data**: Keep fixed labels, scores, or expected outputs alongside generated responses for use in evals.
-- **Generate model responses**: Run a prompt on each row and store the output in a new column, ready for evaluation or comparison.
-- **Add retrieved context**: Fetch relevant chunks from a vector database per row for RAG evaluation or prompt injection.
-- **Classify by category**: Assign topic, sentiment, or intent labels to each row using a model and your predefined categories.
-- **Extract from free text**: Pull specific entities or values from an unstructured column into a clean, structured column.
-
-## How to
-
-Open your dataset and click **+ Add Columns**. Choose **Static** for fixed values or **Dynamic** for computed columns; under Dynamic, pick the method you need.
-
-
-
-
-
- In your dataset, go to the **Data** tab and click **+ Add Columns**. The Add Columns panel opens.
- 
-
-
- Under **Static Columns**, choose the data type: **Text**, **Float**, **Integer**, **Boolean**, **Array**, or **JSON**.
-
-
- Enter a **Column Name** and ensure **Data Type** matches your choice. Click **Create New Column** to add it. You can then fill or edit cells manually.
- 
-
-
-
-
- Choose a dynamic column type below. Configure it, use **Test** to preview, then **Create New Column**.
-
-
-
-
- In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **Run Prompt**.
- 
-
-
- Give the column a name. Build the prompt with messages; use placeholders like {`{{column_name}}`} to pull values from other columns.
- 
-
-
- Select model type (LLM, Text-to-Speech, Speech-to-Text, or Image) and the model. Optionally configure parameters and tools.
-
-
- Set concurrency, then click **Test** to preview outputs. Click **Create New Column** to add the column.
-
-
-
- [Run Prompt in Dataset](/docs/dataset/features/run-prompt) has the full walkthrough.
-
-
-
-
-
- In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **Retrieval**.
-
-
- Name the column. Select the vector database: **Pinecone**, **Qdrant**, or **Weaviate**.
- 
-
-
- Select the **query column**. Add API key/secret. Set **Index Name**, **Namespace**, **Number of Chunks**, and **Query Key**.
- 
-
-
- Set embedding type, model, key to extract, and vector length. Set concurrency, then **Test** and **Create New Column**.
-
-
-
-
-
-
- In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **API Call**.
-
-
- Name the column. Choose **Output Type**: string, object, array, or number.
- 
-
-
- Enter **API URL** and **Method** (GET, POST, PUT, etc.). Add params, headers, and body; use {`{{column_name}}`} to reference column values.
- 
-
-
- Set concurrency. Click **Test** to verify, then **Create New Column**.
-
-
-
-
-
-
- In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **Extract JSON Key**.
-
-
- Name the column. Select the dataset column of type JSON that contains the data.
- 
-
-
- Enter the **JSON key** (path) to extract (e.g. age for a JSON object like {`{"name": "John", "age": 30}`}). Set concurrency, **Test**, then **Create New Column**.
-
-
-
-
-
-
- In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **Extract Entities**.
-
-
- Name the column. Select the column to extract from and enter **instructions** for what to extract.
- 
-
-
- Select the model (API key may be required). Set concurrency, **Test**, then **Create New Column**.
-
-
-
-
-
-
- In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **Classification**.
-
-
- Name the column. Select the column that contains the text to classify.
- 
-
-
- Click **Add Label** and define categories (e.g. Positive, Negative, Neutral). Choose the model and set concurrency.
-
-
- Click **Test** to preview, then **Create New Column**.
-
-
-
-
-
-
- In your dataset, click **+ Add Columns**. Under **Dynamic Columns**, select **Conditional Node**.
-
-
- Name the column. Define **if**, **elif** (optional), and **else** conditions.
- 
-
-
- For each branch, choose an operation (Run Prompt, Retrieval, Extract Entities, Extract JSON, Execute Code, Classification, or API Call) and configure it.
- 
-
-
- Click **Test** to verify, then **Create New Column**.
-
-
-
-
-
-
-
-## Next Steps
-
-
-
- Add individual records or bulk import data rows to your dataset
-
-
- Test and execute prompts against your dataset entries
-
-
- Design and run controlled experiments to compare approaches
-
-
- Add metadata and annotations to enrich your dataset
-
-
- Create another dataset using SDK, file upload, or synthetic generation
-
-
diff --git a/src/pages/docs/dataset/features/add-rows.mdx b/src/pages/docs/dataset/features/add-rows.mdx
deleted file mode 100644
index 6c81c579..00000000
--- a/src/pages/docs/dataset/features/add-rows.mdx
+++ /dev/null
@@ -1,300 +0,0 @@
----
-title: "Adding Data Rows to an Existing Dataset in Future AGI"
-description: "Add data points to an existing dataset manually, from another dataset, Hugging Face, from production traces, or by generating synthetic rows."
----
-
-## About
-
-Add Rows is how you add more data points (rows) to an existing dataset. Each new row gets one cell per column. You either provide the values, copy them from another dataset or source, or generate them (e.g. synthetic or from traces). The dataset's columns stay as they are; only new rows and their cells are created.
-
-## When to use
-
-- **Manual or API data entry**: You have new test cases (e.g. new queries or examples). Add rows with cell values via the UI or API so they become part of the same dataset for run prompt and evals.
-- **Copy from another dataset**: You have rows in a different dataset (or an experiment snapshot) and want them in this one. Add rows from that source with a column mapping so the right fields line up.
-- **Append from Hugging Face**: You want more examples from a Hugging Face dataset. Add rows from that dataset into the current one so you don't re-import from scratch.
-- **Generate more synthetic data**: The dataset was created with synthetic config; you want more rows with the same logic. Add synthetic rows to fill more of the table.
-- **Bring in more production data**: You have new traces/spans in the tracer. Add them to an existing dataset so evals and experiments stay on one dataset.
-
-## How to
-
-Choose how you want to add rows to your dataset:
-
-
-Learn how to [create a new dataset](/docs/dataset/features/create) first if you don't have one yet.
-
-
-
-
-Use the SDK to append rows to an existing dataset.
-
-
- In your app or script, open the dataset you want to add rows to (by name or ID).
-
-
- Define new rows with cells (column name + value), then call the add-rows API.
-
-
-
- ```python Python
- # pip install futureagi
-
- import os
- from fi.datasets import Dataset
- from fi.datasets.types import (
- Cell,
- Column,
- DatasetConfig,
- DataTypeChoices,
- ModelTypes,
- Row,
- SourceChoices,
- )
-
- # Set environment variables
- os.environ["FI_API_KEY"] = ""
- os.environ["FI_SECRET_KEY"] = ""
-
- # Get existing dataset
- config = DatasetConfig(name="Demo-dataset", model_type=ModelTypes.GENERATIVE_LLM)
- dataset = Dataset(dataset_config=config)
- dataset = Dataset.get_dataset_config("Demo-dataset")
-
- # Define columns
- columns = [
- Column(
- name="user_query",
- data_type=DataTypeChoices.TEXT,
- source=SourceChoices.OTHERS
- ),
- Column(
- name="response_quality",
- data_type=DataTypeChoices.INTEGER,
- source=SourceChoices.OTHERS
- ),
- Column(
- name="is_helpful",
- data_type=DataTypeChoices.BOOLEAN,
- source=SourceChoices.OTHERS
- )
- ]
-
- # Define rows
- rows = [
- Row(
- order=1,
- cells=[
- Cell(column_name="user_query", value="What is machine learning?"),
- Cell(column_name="response_quality", value=8),
- Cell(column_name="is_helpful", value=True)
- ]
- ),
- Row(
- order=2,
- cells=[
- Cell(column_name="user_query", value="Explain quantum computing"),
- Cell(column_name="response_quality", value=9),
- Cell(column_name="is_helpful", value=True)
- ]
- )
- ]
-
- try:
- # Add rows to dataset
- dataset = dataset.add_rows(rows=rows)
- print("✓ Data added successfully")
- except Exception as e:
- print(f"Failed to add data: {e}")
-
- ```
-
- ```typescript Typescript
- import { Dataset, DataTypeChoices, createRow, createCell } from "@future-agi/sdk";
-
- process.env["FI_API_KEY"] = "";
- process.env["FI_SECRET_KEY"] = "";
- process.env["FI_BASE_URL"] = "https://api.futureagi.com";
-
- async function main() {
- try {
- const dsName = "Demo-dataset";
-
- // 1) Open the dataset (fetch if it exists, create if not)
- const dataset = await Dataset.open(dsName);
-
- // 2) Define rows
- const rows = [
- createRow({
- cells: [
- createCell({ columnName: "user_query", value: "What is machine learning?" }),
- createCell({ columnName: "response_quality", value: 8 }),
- createCell({ columnName: "is_helpful", value: true }),
- ],
- }),
- createRow({
- cells: [
- createCell({ columnName: "user_query", value: "Explain quantum computing" }),
- createCell({ columnName: "response_quality", value: 9 }),
- createCell({ columnName: "is_helpful", value: true }),
- ],
- }),
- ];
- await dataset.addRows(rows);
- console.log("✓ Data added successfully");
- } catch (err) {
- console.error("Failed to add data:", err);
- }
- }
-
- main();
- ```
-
- ```bash Curl
- curl --request POST \
- --url https://api.futureagi.com/model-hub/develops//add_rows/ \
- --header 'content-type: application/json' \
- --header 'X-Api-Key: ' \
- --header 'X-Secret-Key: ' \
- --data '{
- "rows": [
- {
- "order": 1,
- "cells": [
- {
- "column_name": "user_query",
- "value": "What is machine learning?"
- },
- {
- "column_name": "response_quality",
- "value": 8
- },
- {
- "column_name": "is_helpful",
- "value": true
- }
- ]
- },
- {
- "order": 2,
- "cells": [
- {
- "column_name": "user_query",
- "value": "Explain quantum computing"
- },
- {
- "column_name": "response_quality",
- "value": 9
- },
- {
- "column_name": "is_helpful",
- "value": true
- }
- ]
- }
- ]
- }'
- ```
-
-
- Click [here](https://app.futureagi.com/dashboard/keys) to access API Key and Secret Key.
-
-
-
-
-Add rows using the Add Row option in the dataset view.
-
-
- Open the dataset you want to add rows to from your [dashboard](https://app.futureagi.com/dashboard/develop).
- 
-
-
- Click the "Add Row" option to create one or more new rows. New rows appear at the bottom of the table.
- 
-
-
- Double-click a cell to edit it. Enter values for each column. Repeat for all new rows.
- 
-
-
-
-
-Copy rows from another dataset (or experiment dataset) into this one.
-
-
- From the dataset view, choose the option to add rows from an existing dataset.
- 
-
-
- Select the source dataset (or experiment dataset). Map each source column to a column in the current dataset. Only mapped columns are copied.
- 
-
- | Property | Description |
- | -------- | ----------- |
- | Source dataset | The dataset or experiment dataset to copy rows from |
- | Column mapping | Target column → source column (only mapped columns are copied) |
-
-
- Click "Add" to copy the rows. New rows are appended to the current dataset.
-
-
-
-
-Append rows from a Hugging Face dataset.
-
-
- From the dataset view, choose to add rows from Hugging Face.
- 
-
-
- Search and select the Hugging Face dataset. Choose subset, split, and how many rows to import. Map or confirm columns if required.
- 
-
-
- Start the import. Rows are appended to your dataset and appear in your [dashboard](https://app.futureagi.com/dashboard/develop).
-
-
-
-
-Upload a file (CSV, JSON, JSONL, or Excel) to append rows to your dataset.
-
-
- From the dataset view, choose the option to add rows by uploading a file.
- 
-
-
- Select the dataset you want to add rows to, then upload your file. Column names in the file are matched to existing columns; if the file has new column names, new columns are created on the dataset.
-
- | Property | Description |
- | -------- | ----------- |
- | Dataset | The dataset to append rows to |
- | File | CSV, JSON, JSONL, or Excel file. Column names should match (or will create new columns) |
-
-
- Rows from the file are appended to the dataset. Image and audio values are uploaded to storage. You can run prompt or evals on the updated dataset.
-
-
-
-
-
-
-The number of columns will increase automatically to match the number of columns in the new dataset. And the cells will be None by default.
-
-
-## Next Steps
-
-
-
- Extend your dataset structure with additional data fields
-
-
- Test and execute prompts against your dataset entries
-
-
- Design and run controlled experiments to compare approaches
-
-
- Add metadata and annotations to enrich your dataset
-
-
- Create another dataset using SDK, file upload, or synthetic generation
-
-
\ No newline at end of file
diff --git a/src/pages/docs/dataset/features/annotate.mdx b/src/pages/docs/dataset/features/annotate.mdx
deleted file mode 100644
index 522dc73d..00000000
--- a/src/pages/docs/dataset/features/annotate.mdx
+++ /dev/null
@@ -1,88 +0,0 @@
----
-title: "Adding Annotations to Dataset Rows in Future AGI"
-description: "Annotations are essential for refining datasets, evaluating model outputs, and improving the quality of AI-generated responses."
----
-
-## About
-
-Annotations let you add human labels to dataset rows so you can evaluate model outputs, build training or evaluation data, and improve quality. You create **annotation views** on a dataset: each view defines which columns are shown as context (static fields), which columns hold the content to annotate (response fields), and which **label** (e.g. sentiment, score, free text) is used. Annotators (workspace members you assign) fill in labels per row. For categorical labels, you can optionally use **auto-annotation** to get suggestions based on your existing labels.
-
-## When to use
-
-- **Sentiment analysis**: Categorical labels (e.g. Positive, Negative, Neutral) to measure tone of model outputs.
-- **Factuality check**: Boolean or text labels to validate whether the output is grounded in the source.
-- **Toxicity review**: Categorical labels to flag harmful, biased, or unsafe responses.
-- **Relevance scoring**: Numeric (or star) labels to rate how well the response addresses the query.
-- **Grammar / style edits**: Text labels to provide corrections or rewritten versions.
-- **Prompt comparison**: Categorical or numeric labels to compare responses from different prompt variants.
-
-## How to
-
-
-
- Go to **Datasets** from the dashboard and open the dataset you want to annotate. If you don't have a dataset yet, [create or upload one](/docs/dataset/features/create) first.
- 
-
-
-
- Inside the dataset view, open the **Annotations** tab or button (near the top or side of the data table). This opens the interface for managing annotation views and labels.
- 
-
-
-
- An annotation view defines *what* you annotate and *how*. Click **Create New View**, give the view a **Name** (e.g. "Sentiment Labels", "Fact Check Ratings"), and save. You will configure static fields, response fields, and the label in a later step.
-
-
-
- Labels define the type and possible values for your annotations. Click **Create New Label** if you don't have one. Give the label a **Name** (e.g. "Sentiment", "Accuracy Score") and choose a **Type**: **Categorical** (predefined options, e.g. Positive, Negative, Neutral), **Numeric** (scale with min/max, e.g. 1–5), **Text** (free-form feedback or corrections), **Star** (1–5 stars), or **Thumbs up/down** (pass/fail). Click **Save** to create the label.
- 
- **Auto-annotation (Categorical only):** Enable **Auto-Annotation** and the platform learns from your manual labels and suggests labels for unannotated rows. You can accept or override suggestions.
-
-
-
-
- In the view, connect fields and the label: **Static fields** (columns for context, e.g. user query), **Response fields** (columns to annotate, e.g. model output), **Label** (the label from the previous step). Preview and click **Save**.
-
-
-
- In the annotation view settings, open the **Annotators** section and add workspace members who will annotate in this view.
- 
-
-
-
- Open the annotation view and move through the dataset rows. Click an existing annotation to change it. Changes are saved automatically (or via **Save** if the UI shows it). You can review and override auto-annotation suggestions here as well.
-
-
-
-## Annotation Queues
-
-For structured, multi-annotator annotation campaigns with progress tracking, assignment strategies, and inter-annotator agreement metrics, use **Annotation Queues**. Queues let you organize annotation work across traces, spans, sessions, dataset rows, prototypes, and simulations.
-
-
-
- Learn about annotation queues, labels, and the full annotation workflow
-
-
- Get started with annotation queues in 5 minutes
-
-
-
-## Next Steps
-
-
-
- Add individual records or bulk import data rows to your dataset
-
-
- Extend your dataset structure with additional data fields
-
-
- Test and execute prompts against your dataset entries
-
-
- Design and run controlled experiments to compare approaches
-
-
- Create another dataset using SDK, file upload, or synthetic generation
-
-
diff --git a/src/pages/docs/dataset/features/create.mdx b/src/pages/docs/dataset/features/create.mdx
deleted file mode 100644
index badb83d1..00000000
--- a/src/pages/docs/dataset/features/create.mdx
+++ /dev/null
@@ -1,315 +0,0 @@
----
-title: "Creating a Dataset in Future AGI from Files, SDK, or Traces"
-description: "Create a dataset from CSV, Hugging Face, production traces, or synthetic generation. Use it as the container for prompts, evals, and experiments."
----
-
-## About
-
-Creating a new dataset adds a blank table (or a table filled from a source) under your organization. You get a dataset with a name and optional columns/rows that you can then use for run prompt, evals, experiments, and optimization. The dataset is the container; you can keep editing it after creation.
-
-## When to use
-
-- **Evaluate a prompt or model**: You need a set of inputs and (optionally) expected outputs or scores. Creating a dataset gives you that table so you can run prompts and evals on it.
-- **Reuse production data**: You have traces/spans from your app and want to turn them into eval data. Creating a dataset from Observe turns selected traces into rows.
-- **Import existing data**: You already have test cases in CSV/Excel or on Hugging Face. Creating a dataset from file or Hugging Face imports that data so you don't have to type it in.
-- **Generate test data**: You don't have real data yet but know the kind of examples you need. Creating a [synthetic dataset](/docs/dataset/concept/synthetic-data) generates rows for you.
-- **Branch from an experiment**: You ran an experiment and want to keep that snapshot as a standalone dataset to edit or reuse. Creating a dataset from that experiment copies it into a new dataset.
-
-## How to
-
-Choose how you want to create your dataset:
-
-
-
-Use SDK to import your data to Future AGI.
-
-
- Assign a name to your dataset and click on "Next" to proceed.
-
- 
-
-
- Use the code snippet below to add rows to your dataset.
-
-
-
- ```python Python
- # pip install futureagi
-
- import os
- from fi.datasets import Dataset
- from fi.datasets.types import (
- Cell,
- Column,
- DatasetConfig,
- DataTypeChoices,
- ModelTypes,
- Row,
- SourceChoices,
- )
-
- # Set environment variables
- os.environ["FI_API_KEY"] = ""
- os.environ["FI_SECRET_KEY"] = ""
-
- # Get existing dataset
- config = DatasetConfig(name="my-dataset", model_type= ModelTypes.GENERATIVE_LLM)
- dataset = Dataset(dataset_config=config)
- dataset = Dataset.get_dataset_config("my-dataset")
-
- # Define columns
- columns = [
- Column(
- name="user_query",
- data_type=DataTypeChoices.TEXT,
- source=SourceChoices.OTHERS
- ),
- Column(
- name="response_quality",
- data_type=DataTypeChoices.INTEGER,
- source=SourceChoices.OTHERS
- ),
- Column(
- name="is_helpful",
- data_type=DataTypeChoices.BOOLEAN,
- source=SourceChoices.OTHERS
- )
- ]
-
- # Define rows
- rows = [
- Row(
- order=1,
- cells=[
- Cell(column_name="user_query", value="What is machine learning?"),
- Cell(column_name="response_quality", value=8),
- Cell(column_name="is_helpful", value=True)
- ]
- ),
- Row(
- order=2,
- cells=[
- Cell(column_name="user_query", value="Explain quantum computing"),
- Cell(column_name="response_quality", value=9),
- Cell(column_name="is_helpful", value=True)
- ]
- )
- ]
-
- try:
- # Add columns and rows to dataset
- dataset = dataset.add_columns(columns=columns)
- dataset = dataset.add_rows(rows=rows)
- print("✓ Data added successfully")
-
- except Exception as e:
- print(f"Failed to add data: {e}")
- ```
-
- ```typescript Typescript
- import { Dataset, DataTypeChoices, createRow, createCell } from "@future-agi/sdk";
-
- process.env["FI_API_KEY"] = "";
- process.env["FI_SECRET_KEY"] = "";
-
- async function main() {
- try {
- const dsName = "my-dataset";
-
- // 1) Open the dataset (fetch if it exists, create if not)
- const dataset = await Dataset.open(dsName);
-
- // 2) Define columns
- const columns = [
- { name: "user_query", dataType: DataTypeChoices.TEXT },
- { name: "response_quality", dataType: DataTypeChoices.INTEGER },
- { name: "is_helpful", dataType: DataTypeChoices.BOOLEAN },
- ];
-
- // 3) Define rows
- const rows = [
- createRow({
- cells: [
- createCell({ columnName: "user_query", value: "What is machine learning?" }),
- createCell({ columnName: "response_quality", value: 8 }),
- createCell({ columnName: "is_helpful", value: true }),
- ],
- }),
- createRow({
- cells: [
- createCell({ columnName: "user_query", value: "Explain quantum computing" }),
- createCell({ columnName: "response_quality", value: 9 }),
- createCell({ columnName: "is_helpful", value: true }),
- ],
- }),
- ];
-
- // 4) Add columns and rows
- await dataset.addColumns(columns);
- await dataset.addRows(rows);
- console.log("✓ Data added successfully");
- } catch (err) {
- console.error("Failed to add data:", err);
- }
- }
-
- main();
-
- ```
-
- ```bash cURL
- curl --request POST \
- --url https://api.futureagi.com/model-hub/develops//add_columns/ \
- --header 'X-Api-Key: ' \
- --header 'X-Secret-Key: ' \
- --header 'content-type: application/json' \
- --data '{
- "new_columns_data": [
- {
- "name": "user_query",
- "data_type": "text"
- },
- {
- "name": "response_quality",
- "data_type": "integer"
- },
- {
- "name": "is_helpful",
- "data_type": "boolean"
- }
- ]
- }'
- ```
-
-
- Click [here](https://app.futureagi.com/dashboard/keys) to access API Key and Secret Key.
-
-
-
-
-
-
-
- 
-
-
-
-
-Synthetically generate data and perform experimentations on it.
-
-
-
- Provide basic details about the dataset you want to generate.
-
- 
-
- | Property | Description |
- | -------- | ------------------------------------- |
- | Name | Name of the dataset |
- | Knowledge Base (optional) | Select which knowledge base you want to use. |
- | Description | Describe the dataset you want to generate |
- | Objective (optional) | Use case of the dataset |
- | Pattern (optional) | Style, tone or behavioral traits of the generated dataset |
- | No. of Rows | Row count of the generated dataset (min 10 rows)|
-
-
-
- Define column types and properties
-
- 
-
- | Property | Description |
- | -------- | ------------------------------------- |
- | Column Name | Name of the column |
- | Column Type | Choose the type of the column (available types: text, boolean, integer, float, json, array, datetime) |
-
-
-
- Now add description for each column. Describe in detail what values you want in this column.
- 
-
-
- Click on "Create Dataset" button to generate the dataset. Your synthetic dataset will be generated in a few seconds and will be available in your dataset [dashboard](https://app.futureagi.com/dashboard/develop).
-
- If you are not satisfied with the generated dataset, you can click on "Configure Synthetic Data" button. It will allow you to edit the fields and generate the dataset again.
- 
- 
-
-
-
-
-
-Manually create dataset from scratch.
-
-
-
- To proceed with creating dataset manually from scratch, provide the name you want to assign and the number of columns and rows you want.
- 
- This creates an empty dataset with the name you assigned and empty rows and columns.
- 
-
-
- You can populate the dataset by double-tapping over the empty cell you want to populate. It will open an editor where you can provide the details you want to fill in that cell.
- 
-
-
-
-
-
-
- Search for the dataset you want to import from Hugging Face. You can even refine the search by using flters given on left side.
-
- 
-
-
- Once you have selected the dataset you want to import, click on that dataset and it will open a panel where you can select what subset and split you want to import.
-
- You can also select the number of rows you want to import. By default, it will import all the rows.
- 
-
- Click on "Start Experimenting" button and it will start importing the dataset and you will be able to see it in your dataset [dashboard](https://app.futureagi.com/dashboard/develop).
-
-
-
-
-You can create a subset from an existing dataset.
-
-
- Assign a name to this dataset and choose the existing dataset from the dropdown you want to create a subset from.
- 
- It allows you to import the dataset in two ways:
-
- 1. Import Data: It will only import the original columns from the existing dataset.
- 2. Import Data and Prompt Configuration: Along with original column, it will also import the prompt columns from that dataset.
-
-
- You can choose what columns you want to use from that existing dataset and also you can assign a new name to the columns you want to use.
- 
-
-
-
- Click on "Add" button and it will create a new dataset in your dataset [dashboard](https://app.futureagi.com/dashboard/develop).
-
-
-
-
-
-## Next Steps
-
-
-
- Add individual records or bulk import data rows to your dataset
-
-
- Extend your dataset structure with additional data fields
-
-
- Test and execute prompts against your dataset entries
-
-
- Design and run controlled experiments to compare approaches
-
-
- Add metadata and annotations to enrich your dataset
-
-
diff --git a/src/pages/docs/dataset/features/experiments.mdx b/src/pages/docs/dataset/features/experiments.mdx
deleted file mode 100644
index cd91bea7..00000000
--- a/src/pages/docs/dataset/features/experiments.mdx
+++ /dev/null
@@ -1,149 +0,0 @@
----
-title: "Dataset Experiments: Compare Prompts and Models Side by Side"
-description: "Test different prompt and model combinations on the same dataset. Score outputs with built-in evals and compare results side by side."
----
-
-## About
-
-Experiments give you a structured way to answer questions like: *Which prompt performs better? Which model gives the best results? Does my agent beat my prompt for this task?* You import prompts and agents, run them across multiple model and parameter configurations on the same dataset, score the outputs with evals, and compare results side by side so you can make data-driven decisions instead of guessing.
-
-## When to use
-
-- **Compare prompts and agents**: Pull prompts from the [Prompt](/docs/prompt) section and agents from the [Agent Playground](/docs/agent-playground) into the same experiment and see which produces better outputs.
-- **Compare models and parameters**: Add the same prompt with multiple models, temperatures, or tool configs to compare quality, latency, and cost across configurations.
-- **Validate before rollout**: Test a prompt or agent change on a dataset before promoting it to production.
-- **Optimize with evals**: Attach built-in or custom evals and use scores to rank prompt/agent-model combinations and pick a winner.
-- **Iterate fast**: Stop a long run, edit a single config, or rerun just the failed cells without restarting the whole experiment.
-
-## How to
-
-Experiment creation is a guided three-step flow: **Basic Info → Configuration → Evaluations**. Each step validates before you can move forward, and you can jump back to any completed step to edit it.
-
-
-
- Open the dataset and click the **Experiments** button in the top-right of the dataset dashboard.
- 
-
-
-
- Give the experiment a **name** and pick the **experiment type**.
-
- The name Set up the prompt and model configurations you want to compare. Each configuration becomes a separate column in the experiment grid. is pre-filled with an auto-suggested name based on your dataset. Accept it as-is or overwrite it with your own. Names must be unique within the dataset.
-
- Pick the experiment type that matches the task you're testing:
-
-
-
- Use **LLM** for text generation. You can import prompts *and* agents in the same experiment.
-
-
- Use **TTS** to generate audio from text. Add prompts with different voices, models, and parameters to compare.
-
-
- Use **STT** to transcribe audio. Each prompt configuration must point at a dataset column containing the input audio.
-
-
- Use **Image Generation** to create images from text (or text + image). Compare image models and prompts side by side.
-
-
-
- 
-
-
-
- Set up the prompt and model configurations you want to compare. Each configuration becomes a separate column in the experiment grid.
-
-
-
-
- For LLM experiments, click **Add Prompt/Agents** to import a prompt or agent. You can mix prompts and agents in the same experiment and score them against the same evals.
-
- - **Prompts**: pick a prompt from the [Prompt](/docs/prompt) section, select a published version, then attach **one or more models**. Each (prompt, model) pair becomes its own configuration, so adding three models to one prompt creates three columns to compare. For each model you can tune temperature, max tokens, top-p, response format, and tool config.
- - **Agents**: pick an agent from the [Agent Playground](/docs/agent-playground) and select a published version. The agent's model, tools, and graph are captured at that version, so the run stays reproducible even if the agent is edited later. You don't pick a model again here.
- 
-
-
- For each prompt, write the instructions inline (use `{{column_name}}` to reference dataset columns) and attach one or more **TTS models** (with voice and format settings). Click **+ Add Prompt** to add more prompt entries. Each (prompt, model) pair becomes its own column. Output format is fixed to Audio.
- 
-
-
- For each prompt, write the instructions inline (use `{{column_name}}` to reference dataset columns), pick the dataset column containing the input audio, and attach one or more **STT models**. Click **+ Add Prompt** to add more entries to compare transcription quality.
- 
-
-
- For each prompt, write the instructions inline (use `{{column_name}}` to reference dataset columns) and attach one or more **image models**. Click **+ Add Prompt** to add more entries and compare output quality across models and parameters.
- 
-
-
- Models you've added through Custom Models show up in the model picker for prompt configurations across all experiment types.
-
- See [Custom Models](/docs/evaluation/guides/custom-models) for how to register a custom or self-hosted model.
-
-
-
-
- For prompts, you can also configure **tool calling** with **Auto**, **Required**, or **None**, and add tool definitions the model can invoke.
-
-
-
- The final step has two parts: an optional **base column** and the **evals** you want to score outputs with.
-
- **Compare against baseline (optional)**: pick a column from the dataset to compare model outputs against (typically a ground-truth or existing run-prompt column). Skip it if you don't have a reference output yet; you can still run the experiment, attach evals that don't need a baseline, and add a base column later by editing the experiment.
-
- **Add evaluations**: click **Add Evaluation** and pick from the [built-in eval](/docs/evaluation/builtin) catalog or [create a custom eval](/docs/evaluation/guides/custom-evals). Add as many as you need. Every eval runs on every configuration so the results are directly comparable.
- 
-
- For each eval, map its inputs (e.g. `output`, `input`, `expected`) to the model output or to dataset columns. Mapping is required before the experiment can run.
- 
-
-
-
- Click **Run** to start. The experiment processes every row across every prompt/agent-model configuration in parallel, running the evals on each output as it arrives. The grid streams results live so you can watch progress without waiting for the whole run to finish.
-
-
-
- If you spot a misconfiguration or want to abort, click **Stop** on a running experiment from the Experiments tab. Any in-flight cells are marked as errored, and you can then edit the experiment and rerun without waiting for the full run to complete.
-
-
-
- Use **Rerun Experiment** to re-execute the entire experiment after editing prompts, models, evals, or the base column. Editing is granular: only the configurations you actually changed are re-executed, and results from untouched configurations are preserved.
-
- For more targeted reruns:
-
- - **Rerun a single cell**: hover any output or eval cell in the grid and click the rerun icon. Useful when one row failed transiently or you've tweaked a single configuration.
- - **Rerun a column**: from the column header, choose **Run all cells in the column** or **Run only failed cells in the column**. Failed-only is the fastest way to recover from API hiccups without redoing successful work.
- - **Rerun an eval**: re-execute a single eval across all rows after changing its config or mapping, without re-generating any model outputs.
-
- 
-
-
-
- Open the **Compare** view to see how every configuration performed. Set weights (0-10) for each eval score and for response time, completion tokens, and total tokens. The system normalizes the metrics, computes an overall rating per configuration, and ranks them so the winner is clear. Adjust the weights to match what matters for your use case (e.g. prioritize quality over cost) and the ranking updates in place.
-
-
-
-## Tips
-
-- **Use published versions**: experiments only run published prompt and agent versions. Publish the version you want to test before importing it.
-- **Mix prompts and agents**: an **LLM** experiment can contain prompts and agents side by side, scored against the same evals. Useful when you're deciding whether an agent is worth the extra complexity over a prompt. TTS, STT, and Image experiments accept prompts only.
-- **Failed-only rerun**: when transient failures (rate limits, network blips) leave a few cells errored, use the failed-only rerun on the column to recover them without redoing successful rows.
-
-## Next Steps
-
-
-
- Add individual records or bulk import data rows to your dataset
-
-
- Extend your dataset structure with additional data fields
-
-
- Test and execute prompts against your dataset entries
-
-
- Add metadata and annotations to enrich your dataset
-
-
- Create another dataset using SDK, file upload, or synthetic generation
-
-
diff --git a/src/pages/docs/dataset/features/run-prompt.mdx b/src/pages/docs/dataset/features/run-prompt.mdx
deleted file mode 100644
index 7583abf3..00000000
--- a/src/pages/docs/dataset/features/run-prompt.mdx
+++ /dev/null
@@ -1,126 +0,0 @@
----
-title: "Run Prompt in a Dataset: Generate LLM Columns in Future AGI"
-description: "Add a dynamic column to your dataset by running an LLM, TTS, STT, or image generation model on every row using a prompt with column placeholders."
----
-
-## About
-
-Run Prompt lets you add a new column to your dataset that is filled by a model (LLM, Text-to-Speech, Speech-to-Text, or Image Generation). You define a prompt (messages with placeholders that pull from other columns), pick a model and settings, and the system runs the prompt on each row and writes the model output into that column. The result is a [dynamic column](/docs/dataset/concept/dynamic-column) of responses you can use for evals, comparison, or export.
-
-## When to use
-
-- **Generate answers or text**: Use an LLM to answer questions, summarize, or complete text per row (e.g. a column of questions produces a column of answers).
-- **Produce audio**: Use Text-to-Speech to turn a text column into an audio column (e.g. scripts to voice clips).
-- **Transcribe audio**: Use Speech-to-Text to turn an audio column into a text column for evals or search.
-- **Batch test a prompt**: Run the same prompt across many rows to see how the model behaves and then run evals on the outputs.
-- **Generate images**: Use Image Generation to create images from text (or text + image) per row; the new column stores image URLs.
-- **Structured output**: Use response format (e.g. JSON schema) to get structured fields (object, array) in the new column for downstream use.
-
-## How to
-
-
-
- Click the "Run Prompt" button (e.g. in the top-right or dataset toolbar) to add a new run-prompt column. This creates a dynamic column that will store the model output for each row.
- 
-
-
-
- Enter a name for the prompt. This name is used as the new column name in your dataset. Each row will have one cell in this column holding the model response for that row.
- 
-
-
-
- Select the type of task, then pick the model to use. Models are filtered by type; you need an API key (or custom model) for the chosen provider.
-
-
- Choose **LLM** for text generation (chat). Use for Q&A, summarization, or any text-in, text-out task. Select a chat model from the list; ensure the provider has an API key configured.
- 
-
- Click [here](/docs/evaluation/guides/custom-models) to learn how to create custom models.
-
-
-
- Choose **Text-to-Speech** to generate audio from text. The prompt output column will store audio (e.g. URLs). You can configure voice and format for supported TTS models.
- 
-
- Click [here](/docs/evaluation/guides/custom-models) to learn how to create custom models.
-
-
-
- Choose **Speech-to-Text** to transcribe audio into text. Use when a column contains audio; the model output will be text in the new column.
- 
-
- Click [here](/docs/evaluation/guides/custom-models) to learn how to create custom models.
-
-
-
- Choose **Image Generation** to create images from text (or image + text) prompts. The prompt output column will store image URLs. Select an image-generation model and ensure the provider has an API key configured.
- 
-
- Click [here](/docs/evaluation/guides/custom-models) to learn how to create custom models.
-
-
-
-
-
-
- Define the prompt as a list of messages with roles:
-
- - **System** (optional): Instructions that guide the model's behavior and set context.
- - **User** (required): The main input message. This role is required for the prompt to work.
-
- Use `{{column_name}}` placeholders to pull values from other columns. At runtime, these are replaced by the cell value for each row.
-
- **Example:**
- ```
- System: You are a helpful assistant that summarizes content.
-
- User: Please summarize the following text: {{article_text}}
- ```
-
- **JSON dot notation**: For JSON columns, access nested fields directly:
- ```
- User: Based on this prompt: {{config.prompt}}, generate a response that addresses {{config.topic}}
- ```
-
- `{{config.prompt}}` accesses the `prompt` field within the `config` JSON column.
-
-
-
- Adjust model parameters such as temperature, max tokens, top_p, and other settings to fine-tune the model's behavior according to your needs.
-
-
-
- Add tools or functions that the model can use during execution. This enables the model to perform specific actions or access external capabilities.
-
-
-
- Set the concurrency level to control how many prompt executions run in parallel. Higher concurrency speeds up processing but may consume more resources.
-
-
-
- Click the "Run" button to execute the prompt across your dataset. The responses will be generated and saved as a new dynamic column in your dataset.
-
-
-
-
-
-## Next Steps
-
-
-
- Add individual records or bulk import data rows to your dataset
-
-
- Extend your dataset structure with additional data fields
-
-
- Design and run controlled experiments to compare approaches
-
-
- Add metadata and annotations to enrich your dataset
-
-
- Create another dataset using SDK, file upload, or synthetic generation
-
-
diff --git a/src/pages/docs/dataset/guides/add-columns.mdx b/src/pages/docs/dataset/guides/add-columns.mdx
new file mode 100644
index 00000000..92f0f404
--- /dev/null
+++ b/src/pages/docs/dataset/guides/add-columns.mdx
@@ -0,0 +1,67 @@
+---
+title: "Add columns"
+description: "Create a static column for values you set yourself, or a dynamic one that computes them, from the same panel."
+---
+
+Every column starts in the same panel, whether it holds values you type in or values a method computes for you. Open it, pick a type, name the column, and either save it right away or test it first.
+
+From the dataset's **Data** tab, click **Add Column**. The Add Columns panel opens with a filter on the left: **All**, **Static Columns**, or **Dynamic Columns**. Pick a filter (or search by name), then click the type you want. A [static column](/docs/dataset/concepts/static-and-dynamic-columns) creates immediately once you name it; a dynamic column opens a fuller form for the method's settings, and lets you test it before you commit.
+
+## Add a static column
+
+Static columns are the fastest path: pick a data type, name it, and it's in the grid ready to fill in.
+
+
+
+ Filter to **Static Columns** and click the data type you want, for example **Text**. The full set of types, and what each one stores, is in [Limits & Data Types](/docs/dataset/reference/limits-and-data-types).
+
+
+ A small panel opens with a **Column name** field. Name it `reviewer_notes` and click **Add Column**.
+
+
+
+The column appears in the grid right away, empty, ready for you to fill in cell by cell.
+
+## Add a dynamic column
+
+Dynamic columns point at a method instead of holding values you type. The example here uses **Classification**, which reads another column's text and sorts each row into one of the labels you define. Run Prompt is a dynamic column too, but it gets its own walkthrough in [Run a prompt on every row](/docs/dataset/guides/run-a-prompt-on-every-row); every other method is cataloged in [Dynamic column methods](/docs/dataset/reference/dynamic-column-methods).
+
+
+
+ Filter to **Dynamic Columns** and click **Classification**.
+
+
+ Name the column `query_topic`, then select the column to classify: `user_query`.
+
+
+ Add each label the model can choose from, for example `Billing`, `Technical`, and `Account`.
+
+
+ Choose the model that runs the classification, and set how many rows it processes at once.
+
+
+ Click **Test** to preview the labels it would assign, without saving anything yet. Once it looks right, click **Create New Column**.
+
+
+
+Create New Column starts Classification on every row, filling `query_topic` in as it works through the dataset.
+
+## Validation you'll hit
+
+- Column names cap at 255 characters
+- A name that's already used in this dataset is rejected
+- Two columns in the same request can't share a name, which only comes up when you add more than one column at once, for example through the SDK
+
+## Dive deeper
+
+
+
+ Walk the Run Prompt method end to end
+
+
+ Every other method you can point a dynamic column at
+
+
+ Rename, retype, or delete a column after it's in
+
+
diff --git a/src/pages/docs/dataset/guides/add-rows.mdx b/src/pages/docs/dataset/guides/add-rows.mdx
new file mode 100644
index 00000000..31f422f7
--- /dev/null
+++ b/src/pages/docs/dataset/guides/add-rows.mdx
@@ -0,0 +1,182 @@
+---
+title: "Add rows"
+description: "Five ways to add new examples to a dataset you've already created."
+---
+
+Adding rows brings more examples into a [dataset](/docs/dataset/concepts/understanding-datasets) that already exists. If you don't have a dataset yet, start with [Create a dataset](/docs/dataset/guides/create-a-dataset). Columns stay put unless imported data brings a name the dataset doesn't already have. Whichever path you pick, new rows always land after whatever's already in the dataset, appended in the order you add them.
+
+## Add rows using the SDK
+
+Use this when you're scripting the setup or pushing rows from your own pipeline.
+
+
+
+```python Python
+# pip install futureagi
+
+import os
+from fi.datasets import Dataset
+from fi.datasets.types import Cell, Row
+
+os.environ["FI_API_KEY"] = ""
+os.environ["FI_SECRET_KEY"] = ""
+
+dataset = Dataset.get_dataset_config("support-agent-eval")
+
+rows = [
+ Row(cells=[
+ Cell(column_name="user_query", value="How do I reset my password?"),
+ Cell(column_name="response_quality", value=7),
+ Cell(column_name="is_helpful", value=True),
+ ]),
+ Row(cells=[
+ Cell(column_name="user_query", value="What's your refund policy?"),
+ Cell(column_name="response_quality", value=9),
+ Cell(column_name="is_helpful", value=True),
+ ]),
+]
+
+dataset = dataset.add_rows(rows=rows)
+```
+
+```typescript Typescript
+import { Dataset, createRow, createCell } from "@future-agi/sdk";
+
+process.env["FI_API_KEY"] = "";
+process.env["FI_SECRET_KEY"] = "";
+
+async function main() {
+ const dataset = await Dataset.open("support-agent-eval", { createIfMissing: false });
+
+ const rows = [
+ createRow({
+ cells: [
+ createCell({ columnName: "user_query", value: "How do I reset my password?" }),
+ createCell({ columnName: "response_quality", value: 7 }),
+ createCell({ columnName: "is_helpful", value: true }),
+ ],
+ }),
+ createRow({
+ cells: [
+ createCell({ columnName: "user_query", value: "What's your refund policy?" }),
+ createCell({ columnName: "response_quality", value: 9 }),
+ createCell({ columnName: "is_helpful", value: true }),
+ ],
+ }),
+ ];
+
+ await dataset.addRows(rows);
+}
+
+main();
+```
+
+```bash Curl
+curl --request POST \
+ --url https://api.futureagi.com/model-hub/develops//add_rows/ \
+ --header 'content-type: application/json' \
+ --header 'X-Api-Key: ' \
+ --header 'X-Secret-Key: ' \
+ --data '{
+ "rows": [
+ {
+ "cells": [
+ { "column_name": "user_query", "value": "How do I reset my password?" },
+ { "column_name": "response_quality", "value": 7 },
+ { "column_name": "is_helpful", "value": true }
+ ]
+ },
+ {
+ "cells": [
+ { "column_name": "user_query", "value": "What'\''s your refund policy?" },
+ { "column_name": "response_quality", "value": 9 },
+ { "column_name": "is_helpful", "value": true }
+ ]
+ }
+ ]
+}'
+```
+
+
+
+A `Row` accepts an `order`, but new rows are always appended after the last existing one, whatever order you pass, so it isn't how you control placement. See the [Datasets SDK](/docs/sdk/datasets) page for the full `Dataset` class.
+
+Get your API key and secret key [here](https://app.futureagi.com/dashboard/keys).
+
+Every drawer path below starts the same way: on the dataset's **Data** tab, click **Add Row** to open the drawer, which has a tile for each way to add rows.
+
+## Add a row from the grid
+
+Use this for typing in one or two examples by hand.
+
+
+
+ Pick **Add empty row** and choose how many to add.
+
+
+ New rows appear at the bottom of the table with empty cells. Double-click a cell to enter a value, and repeat for each row.
+
+
+
+Instead of starting blank, you can also duplicate rows you already have: select them in the grid, click **Duplicate** in the toolbar, then set the number of copies.
+
+## Copy rows from another dataset or experiment
+
+Use this when the rows you need already exist somewhere else.
+
+
+
+ In the Add Row drawer, select **Add from existing model dataset or experiment**, then pick the source: another dataset, or an experiment snapshot.
+
+
+ Map each source column to a column in this dataset. Only mapped columns copy over; anything left unmapped in the source is skipped.
+
+
+
+## Import rows from Hugging Face
+
+Use this to pull in a public dataset instead of typing examples by hand.
+
+
+
+ In the Add Row drawer, select **Import from Hugging Face**, then search for the dataset and pick the subset next to the split. Set how many rows to import.
+
+
+ Start the import. Each Hugging Face feature is matched to a column by name; a name that doesn't exist yet becomes a new column, backfilled with empty cells on the rows that already existed.
+
+
+
+## Add rows from a file
+
+Use this for a CSV, Excel, JSON or JSONL export you already have.
+
+
+
+ In the Add Row drawer, select **Upload a file (JSONl/ JSON/ CSV)**, then upload it.
+
+
+ Column names in the file are matched to existing columns by name. A name that isn't already a column gets created, backfilled with empty cells on the rows that already existed.
+
+
+
+## Caps you'll hit
+
+- Adding empty rows from the grid: the picker tops out at 10 at a time, up to 100 per request via the API
+- Duplicating a row: at most 100 copies
+- Uploading a file: 25 MB, restricted to `.csv`, `.xls`, `.xlsx`, `.json`, `.jsonl`
+
+The full set of dataset and column limits is in [Limits & Data Types](/docs/dataset/reference/limits-and-data-types).
+
+## Dive deeper
+
+
+
+ Extend the dataset with a static or dynamic column
+
+
+ Turn a prompt into a column of model output
+
+
+ Rename, edit, or delete what's already in the grid
+
+
diff --git a/src/pages/docs/dataset/guides/create-a-dataset.mdx b/src/pages/docs/dataset/guides/create-a-dataset.mdx
new file mode 100644
index 00000000..8abce2db
--- /dev/null
+++ b/src/pages/docs/dataset/guides/create-a-dataset.mdx
@@ -0,0 +1,172 @@
+---
+title: "Create a dataset"
+description: "Six ways to create a dataset and get data into it."
+---
+
+A dataset starts as just a name in your organization; everything else comes from how you fill it. Future AGI gives you six ways to do that, each landing you on the same [dataset](/docs/dataset/concepts/understanding-datasets) table of rows and columns you can keep editing afterward.
+
+Every method starts the same way: from the dataset list (**Dataset** in the left nav), click **Add Dataset**. That opens a panel with six tiles:
+
+| Method | Reach for it when |
+| --- | --- |
+| Add data using SDK | You're scripting the setup or pushing rows from your own pipeline |
+| Upload a file (JSON, CSV) | You already have test cases in a CSV, Excel, or JSON export |
+| Create Synthetic Data | You don't have real data yet, but know the shape of the examples you need |
+| Add datasets Manually | You're hand-building a small set and want an empty grid to fill in |
+| Import from HuggingFace | The data you need already exists as a Hugging Face dataset |
+| Add from existing model dataset or experiment | You want to branch off a dataset or experiment you already have |
+
+Whichever method you pick, the dataset's name must be unique inside your organization. A name that's already taken is rejected. For the full set of dataset and column limits, see [Limits & Data Types](/docs/dataset/reference/limits-and-data-types).
+
+## Add data using the SDK
+
+For scripting dataset creation, or when you'd rather write rows in code than click through the grid.
+
+In the Add dataset panel, pick **Add data using SDK** and name the dataset. Future AGI creates an empty dataset and drops you on its Data tab, which shows the dataset's name, ID, API key, and secret key alongside a ready-to-run code snippet. Copy the snippet below and run it against your dataset to add [columns](/docs/dataset/concepts/static-and-dynamic-columns) and rows.
+
+An empty dataset starts with at most 10 rows; add the rest with the SDK.
+
+
+
+```python Python
+# pip install futureagi
+
+import os
+from fi.datasets import Dataset
+from fi.datasets.types import Cell, Column, DataTypeChoices, Row, SourceChoices
+
+os.environ["FI_API_KEY"] = ""
+os.environ["FI_SECRET_KEY"] = ""
+
+# Get the dataset you just created
+dataset = Dataset.get_dataset_config("support-agent-eval")
+
+# Define columns
+columns = [
+ Column(name="user_query", data_type=DataTypeChoices.TEXT, source=SourceChoices.OTHERS),
+ Column(name="response_quality", data_type=DataTypeChoices.INTEGER, source=SourceChoices.OTHERS),
+ Column(name="is_helpful", data_type=DataTypeChoices.BOOLEAN, source=SourceChoices.OTHERS),
+]
+
+# Define rows
+rows = [
+ Row(order=1, cells=[
+ Cell(column_name="user_query", value="What is machine learning?"),
+ Cell(column_name="response_quality", value=8),
+ Cell(column_name="is_helpful", value=True),
+ ]),
+ Row(order=2, cells=[
+ Cell(column_name="user_query", value="Explain quantum computing"),
+ Cell(column_name="response_quality", value=9),
+ Cell(column_name="is_helpful", value=True),
+ ]),
+]
+
+dataset = dataset.add_columns(columns=columns)
+dataset = dataset.add_rows(rows=rows)
+```
+
+```typescript Typescript
+import { Dataset, DataTypeChoices, createRow, createCell } from "@future-agi/sdk";
+
+process.env["FI_API_KEY"] = "";
+process.env["FI_SECRET_KEY"] = "";
+
+async function main() {
+ // Get the dataset you just created
+ const dataset = await Dataset.open("support-agent-eval");
+
+ // Define columns
+ const columns = [
+ { name: "user_query", dataType: DataTypeChoices.TEXT },
+ { name: "response_quality", dataType: DataTypeChoices.INTEGER },
+ { name: "is_helpful", dataType: DataTypeChoices.BOOLEAN },
+ ];
+
+ // Define rows
+ const rows = [
+ createRow({
+ cells: [
+ createCell({ columnName: "user_query", value: "What is machine learning?" }),
+ createCell({ columnName: "response_quality", value: 8 }),
+ createCell({ columnName: "is_helpful", value: true }),
+ ],
+ }),
+ createRow({
+ cells: [
+ createCell({ columnName: "user_query", value: "Explain quantum computing" }),
+ createCell({ columnName: "response_quality", value: 9 }),
+ createCell({ columnName: "is_helpful", value: true }),
+ ],
+ }),
+ ];
+
+ await dataset.addColumns(columns);
+ await dataset.addRows(rows);
+}
+
+main();
+```
+
+```bash Curl
+curl --request POST \
+ --url https://api.futureagi.com/model-hub/develops//add_columns/ \
+ --header 'X-Api-Key: ' \
+ --header 'X-Secret-Key: ' \
+ --header 'content-type: application/json' \
+ --data '{
+ "new_columns_data": [
+ {"name": "user_query", "data_type": "text"},
+ {"name": "response_quality", "data_type": "integer"},
+ {"name": "is_helpful", "data_type": "boolean"}
+ ]
+ }'
+```
+
+
+
+See the [Datasets SDK reference](/docs/sdk/datasets) for the full `Dataset` API.
+
+## Upload a file
+
+For bringing in test cases you already have as a file, instead of typing them in.
+
+In the Add dataset panel, pick **Upload a file (JSON, CSV)** and name the dataset. Drop or browse to your file: accepted formats are `.csv`, `.xls`, `.xlsx`, `.json`, and `.jsonl`, up to 25 MB. The dataset appears on your list right away. Future AGI processes the file in the background with a visible progress state until it's done.
+
+## Create synthetic data
+
+For when you don't have real data yet, but know the shape of the examples you need.
+
+In the Add dataset panel, pick **Create Synthetic Data** and name the dataset. From there, Future AGI walks you through describing the schema and generates rows for you. See [Synthetic Data](/docs/dataset/concepts/synthetic-data) if you want to regenerate later.
+
+## Add a dataset manually
+
+For hand-building a small dataset from scratch when you already know its shape.
+
+In the Add dataset panel, pick **Add datasets Manually**, name the dataset, and choose how many rows and how many columns to start with, up to 100 of each. Future AGI creates the dataset with that many empty rows and columns, ready for you to fill in.
+
+## Import from Hugging Face
+
+For pulling in a Hugging Face dataset instead of typing test cases by hand.
+
+In the Add dataset panel, pick **Import from HuggingFace** and paste the Hugging Face dataset ID. Click **Load Dataset**, then pick the **Subset** and **Split** you want. Name the new dataset to finish. Only the first 100 rows of the source are ingested.
+
+## Add from an existing dataset or experiment
+
+For branching off a dataset or experiment you already have.
+
+In the Add dataset panel, pick **Add from existing model dataset or experiment** and choose the dataset or experiment you want to copy from. Choose whether to bring over **Import Data** or **Import data and prompt configuration**, then select which columns to include. Name the new dataset to finish.
+
+## Dive deeper
+
+
+
+ Put more records into a dataset that already exists
+
+
+ Extend a dataset with a static or dynamic column
+
+
+ Turn a prompt into a column of model output
+
+
diff --git a/src/pages/docs/dataset/guides/manage-datasets.mdx b/src/pages/docs/dataset/guides/manage-datasets.mdx
new file mode 100644
index 00000000..b496cc7d
--- /dev/null
+++ b/src/pages/docs/dataset/guides/manage-datasets.mdx
@@ -0,0 +1,78 @@
+---
+title: "Manage datasets"
+description: "Find, duplicate, export, and clean up datasets and their columns once the data is already in"
+---
+
+Once a dataset has data in it, the work shifts from adding rows to keeping the table itself in order: finding the right dataset, making a copy, pulling data out, fixing a column, or getting rid of what you no longer need. This guide covers all of that on an existing dataset. For putting data in, see [Create a dataset](/docs/dataset/guides/create-a-dataset), [Add rows](/docs/dataset/guides/add-rows), and [Add columns](/docs/dataset/guides/add-columns).
+
+## Find a dataset in the list
+
+- Only **Dataset Name** and **Datapoints** are sortable: click either header to reorder the list by it
+- Use the search box above the table to filter by name
+- If nothing matches, the list shows **No datasets found** instead of an empty table
+
+The list is paginated; see [Limits & Data Types](/docs/dataset/reference/limits-and-data-types) for the page size.
+
+## Act on several datasets at once
+
+Select one or more datasets with the row checkboxes and a bulk action bar takes over the toolbar, showing **{'{n}'} Selected** alongside **Delete** and **Cancel**. Select exactly one dataset and **Duplicate** joins the bar; select two or more and **Duplicate** is replaced by **Compare**, which opens a **Select Base Columns** drawer where you pick the one column the selected datasets share. Cancel clears the selection without doing anything.
+
+
+Every control that changes a dataset (duplicate, edit, and delete) is gated on your dataset permission. In the bulk action bar, this means the controls show up disabled rather than doing nothing when clicked. In the grid, Edit Column Name, Edit Column Type, and Delete Column don't appear in the column header menu at all for a viewer, and cells simply stop being editable. Downloading isn't role-gated, but the download button is disabled when the dataset has no data, or when a synthetic dataset is still processing.
+
+
+## Duplicate a dataset
+
+Check the dataset's row checkbox, then click **Duplicate** in the bulk action bar to open the **Duplicate Dataset** dialog. Enter a name for the copy in **Enter Dataset Name**, for example `support-agent-eval-copy`, then click **Create**. The dialog validates the name before it lets you proceed, and a success toast confirms once the copy exists. Click **Cancel** to back out without duplicating anything.
+
+The duplicate is a separate dataset from the moment it's created: editing it doesn't touch the original, and deleting one doesn't touch the other. It isn't a full copy, though: duplicating only carries over rows and [static columns](/docs/dataset/concepts/static-and-dynamic-columns). Dynamic columns, and the computed values in them, don't come across, so a duplicated dataset can have fewer columns than the one it was made from.
+
+## Export a dataset
+
+Click the download icon in the dataset's toolbar to export it. A **Download has been started...** toast appears immediately, followed by **Dataset downloaded successfully** once the file is ready.
+
+## Edit data in the grid
+
+Inside a dataset, the grid supports these changes directly, without leaving the Data tab. Renaming a column, changing its data type, and deleting it all live in the column header menu, under **Edit Column Name**, **Edit Column Type**, and **Delete Column**.
+
+| Action | What it does |
+|---|---|
+| Rename a column (`response_quality` to `quality_score`, for example) | The cell values stay the same, but SDK and cURL calls that reference the old name break |
+| Change a column's data type | Reinterprets how the column's stored values are treated |
+| Edit a cell in a static column | Overwrites that one cell's value; audio and persona cells can't be edited this way |
+| Delete a column | Removes the column and every cell in it |
+
+
+A column has no identifier besides its name, so a rename changes what your integrations have to send. `add_rows` and other SDK or cURL calls key each cell by `column_name`; if a call still references `response_quality` after you rename it to `quality_score`, that call fails until it's updated.
+
+
+To delete rows, select them with their row checkboxes and confirm the delete action in the grid toolbar.
+
+Cells in a dynamic column can't be edited at all: the column is managed by whatever produces it.
+
+## Delete a dataset
+
+Select one or more datasets with the row checkboxes, then click **Delete** in the bulk action bar. The dialog title switches between **Delete Dataset** and **Delete Datasets** depending on how many you selected. Confirm with **Delete**, or back out with **Cancel**. A success toast confirms once it's done.
+
+Bulk delete is capped at 50 datasets per request. Deleting more than that means running the action in batches.
+
+## What deleting actually removes
+
+Deletes here are final: once you delete a row, a column, or a dataset, it's gone. Deleting rows removes the selected rows and every cell in them. Deleting a column removes the column and every cell in it.
+
+
+Deleting a dynamic column also deletes the producer behind it, whether that's a prompt run or an eval, and anything else that was derived from it. It isn't just the column that disappears.
+
+
+Deleting a dataset also deletes its [experiments](/docs/dataset/guides/run-an-experiment).
+
+## Dive deeper
+
+
+
+ What a column's producer is, and why deleting one takes it along
+
+
+ Exact numbers for every cap on this page
+
+
diff --git a/src/pages/docs/dataset/guides/run-a-prompt-on-every-row.mdx b/src/pages/docs/dataset/guides/run-a-prompt-on-every-row.mdx
new file mode 100644
index 00000000..adad9356
--- /dev/null
+++ b/src/pages/docs/dataset/guides/run-a-prompt-on-every-row.mdx
@@ -0,0 +1,88 @@
+---
+title: "Run a prompt on every row"
+description: "Turn a prompt into a column of model output, one response per row."
+---
+
+**Run Prompt** fills a new [dynamic column](/docs/dataset/concepts/static-and-dynamic-columns) by running a prompt against every row of a dataset that already exists. You write the prompt once, referencing other columns as inputs, and Future AGI runs it row by row until the whole column is filled. You'll need a dataset with the input columns your prompt will reference.
+
+
+
+ On a dataset's **Data** tab, click **Run Prompt**. This starts a new column and opens the panel where you build the prompt and pick a model.
+
+
+
+ In the **Name** field (placeholder "Prompt Name"), name the new column `model_response` for this example. It's the first field in the panel, above the model type options, and it becomes the name of the column every row's response lands in.
+
+
+
+ Run Prompt supports four kinds of models, each shaped the same way: pick a type, then pick the specific model from that type's list.
+
+ | Model type | Input | Output |
+ | --- | --- | --- |
+ | LLM | The prompt you build next | Text |
+ | Text-to-Speech | A text column referenced in the prompt | Audio |
+ | Speech-to-Text | An audio column | Transcribed text |
+ | Image Generation | A single image prompt | An image |
+
+ LLM prompts are the message-based kind covered next. For LLM models, don't see the model you need? [Register a custom model](/docs/evaluation/guides/custom-models) and it joins the same list.
+
+
+
+ An LLM prompt is a list of messages with roles:
+
+ - **System** (optional): instructions that set the model's behavior and context
+ - **User** (required): the input message, built from your dataset's columns
+
+ Use `{{column_name}}` inside a message to pull that column's value for the current row. Take a dataset with a `user_query` column and a `customer_context` JSON column:
+
+ **System**
+ ```
+ You are a support assistant that helps resolve customer tickets.
+ ```
+
+ **User**
+ ```
+ A customer asked: {{user_query}}. Their plan is {{customer_context.plan}}. Write a helpful response.
+ ```
+
+ `{{user_query}}` pulls that row's plain text value. `{{customer_context.plan}}` uses dot notation to reach the `plan` field inside the `customer_context` JSON column, without pulling in the rest of that column's value.
+
+ Text-to-Speech, Speech-to-Text, and Image Generation prompts are simpler, since each is a single input instead of a message list:
+
+ - **Text-to-Speech**: in the Prompt Input box, reference the text column to speak, for example `{{script_text}}`, and choose a Voice, which is required
+ - **Speech-to-Text**: pick a column in the Voice Input section's Column dropdown, which lists your audio columns; selecting one fills the message for you
+ - **Image Generation**: write the prompt describing the image directly in the Image Prompt field
+
+
+
+ Set how many rows run at once, from 1 to 10. It defaults to 5.
+
+
+
+ Adjust generation parameters such as temperature, top P, max tokens, presence and frequency penalty, and response format, if the defaults don't fit your prompt, from the options button beside **Select Model**.
+
+ Attach tools the model can call while it runs, if your prompt needs them, in the **Tool Configuration** accordion above **Concurrency**.
+
+
+
+ Click **Run**. Future AGI works through the dataset row by row and writes each response into the new column. Watch a row's cell to see it complete; the column is done once every cell has filled.
+
+
+
+## What lands in the column
+
+While a row's call is in flight, its cell shows a loading placeholder until the response lands. If the call fails, its cell shows an error. Otherwise the cell fills with the response, and each LLM cell also records its token counts and response time.
+
+## Dive deeper
+
+
+
+ Add a static or dynamic column from the Data tab
+
+
+ Every other producer a dynamic column can point at
+
+
+ Compare prompts and models against each other using evals
+
+
diff --git a/src/pages/docs/dataset/guides/run-an-experiment.mdx b/src/pages/docs/dataset/guides/run-an-experiment.mdx
new file mode 100644
index 00000000..4b87234d
--- /dev/null
+++ b/src/pages/docs/dataset/guides/run-an-experiment.mdx
@@ -0,0 +1,67 @@
+---
+title: "Run an experiment"
+description: "Set up an experiment, run it across your dataset, and use Choose winner to pick the best configuration."
+---
+
+An **experiment** runs every [prompt or agent](/docs/prompt) and model combination you set up against the same [dataset](/docs/dataset/concepts/understanding-datasets), scored by the [evals](/docs/evaluation) you attach, so you can compare configurations side by side instead of testing them one at a time.
+
+## The experiment grid
+
+One experiment lays your dataset's rows against every configuration you add. A **configuration** pairs one prompt or agent with one model, so attaching three models to the same prompt gives you three configurations, one column each. Every eval you attach scores every configuration against the same rows, which is what makes the columns comparable.
+
+ GRID{{"Experiment grid"}}
+ PA1["Prompt or agent"] --> CFG1["Configuration A"]
+ MD1["Model"] --> CFG1
+ PA2["Prompt or agent"] --> CFG2["Configuration B"]
+ MD2["Model"] --> CFG2
+ CFG1 --> GRID
+ CFG2 --> GRID
+ EV["Evals"] -->|"scores every column"| GRID`} />
+
+## Build the experiment
+
+Click **Experiment** on the dataset to open the creation flow. It's a three-step form.
+
+
+
+ Name the experiment and choose its type: **LLM**, **TTS**, **STT**, or **Image**. The type decides the output format and which models you can attach.
+
+
+ Add the prompts or agents you want to compare and attach a model to each. Every prompt/agent-model pair becomes its own configuration column. LLM experiments can mix prompts and agents in the same run and attach tools to a prompt, useful for deciding whether an agent earns its extra complexity over a plain prompt; TTS, STT, and Image experiments take prompts only.
+
+
+ Optionally pick a column to compare outputs against as a baseline, then add the evals that will score every configuration.
+
+
+
+Click **Run Experiment**, and every row runs against every configuration, with each output scored by your evals as it comes in.
+
+
+An eval you add after the run sits queued for a few seconds before it starts scoring.
+
+
+## Stop and rerun
+
+Each experiment stops and reruns independently. Stop a running one without touching the others in the dataset, and rerun a completed, failed, or cancelled one later without setting it up again, though rerunning overwrites its existing results. Select more than one experiment at a time to rerun or delete them together.
+
+## Choose a winner
+
+The experiment summary already lists every configuration. Once every configuration has a score, click **Choose winner** to open Winner Settings, where you set the importance of Average Response Time, Completion tokens, Total tokens, and each eval. Click **Save & Run** and the summary marks the winning configuration.
+
+## Tips
+
+- **Failed-only rerun**: when transient failures (rate limits, network blips) leave a few cells errored, use the failed-only rerun on the column to recover them without redoing successful rows
+
+## Dive deeper
+
+
+
+ Add human labels to rows once the experiment tells you where to look
+
+
+ Duplicate, export, or clean up a dataset after you're done experimenting
+
+
diff --git a/src/pages/docs/dataset/index.mdx b/src/pages/docs/dataset/index.mdx
index ab2b51a3..19097f24 100644
--- a/src/pages/docs/dataset/index.mdx
+++ b/src/pages/docs/dataset/index.mdx
@@ -1,65 +1,36 @@
---
-title: "Future AGI Datasets: Evaluation and Experimentation Layer"
-description: "Structured tables of examples for prompts, evaluations, and experiments. Create from file uploads, SDK, production traces, or synthetic generation."
+title: "Overview"
+description: "What a dataset is made of, where the data comes from, and where to go next"
---
-## About
+## What is a Dataset?
-Datasets are the core data layer for evaluation and experimentation in Future AGI. Each dataset is a table with columns (e.g. "user query", "expected answer", "score"), rows (one row per example), and cells (the value in each column for each row).
+A **dataset** is a table of examples. [Prompts](/docs/prompt) and [evals](/docs/evaluation) run against it and write their results back as new columns. [Experiments](/docs/dataset/guides/run-an-experiment) and [optimization](/docs/optimization) run on the same rows in their own tabs. You reach it from **Dataset** in the left nav.
-Datasets are the single source of truth that prompts, evaluations, experiments, and optimizations run on. You can create them from file uploads, the SDK, observed production traces, or synthetic generation.
+## Columns, rows, and cells
-
+Each dataset is a grid: **columns** define what you're capturing (a query, an expected answer, a score), **rows** are the individual examples, and a **cell** holds the value where a row meets a column. A column's values come from you directly, or get filled in automatically by running something against the dataset. See [Static & Dynamic Columns](/docs/dataset/concepts/static-and-dynamic-columns) for the difference.
-## Column Types
+## Where the data comes from
-Datasets support two types of columns:
+- **[File upload](/docs/dataset/guides/create-a-dataset#upload-a-file)**: bring in a CSV, JSON, or JSONL file
+- **[The SDK](/docs/dataset/guides/create-a-dataset#add-data-using-the-sdk)**: push rows from your own code
+- **[Synthetic generation](/docs/dataset/guides/create-a-dataset#create-synthetic-data)**: describe a schema and get realistic rows back
+- **[Hugging Face](/docs/dataset/guides/create-a-dataset#import-from-hugging-face)**: import an existing dataset by name
+- **[An existing dataset or experiment](/docs/dataset/guides/create-a-dataset#add-from-an-existing-dataset-or-experiment)**: branch off data you already have in Future AGI
+- **[Observe](/docs/observe) traces**: turn real production traffic into rows
+- **[Manual entry](/docs/dataset/guides/create-a-dataset#add-a-dataset-manually)**: add rows and columns by hand
-- **Static columns**: Data you add directly, either manually, via file upload, or through the SDK. These hold your inputs, expected outputs, ground truth labels, or any fixed data.
-- **Dynamic columns**: Generated on-the-fly by running a prompt, evaluation, or model against your dataset rows. For example, running GPT-4o on every row creates a dynamic column with the model's responses.
+## Dive deeper
-This distinction matters because dynamic columns let you add model outputs, evaluation scores, and computed fields to your dataset without duplicating data.
-
-## How Datasets Connect to Other Features
-
-- **Evaluation**: Run 70+ built-in metrics across your dataset rows to score model outputs. Results are stored as new columns. [Learn more](/docs/evaluation)
-- **Experiments**: Compare two prompts or models by running both against the same dataset and comparing scores side by side. [Learn more](/docs/dataset/features/experiments)
-- **Optimization**: Use datasets as the training ground for prompt optimization algorithms. [Learn more](/docs/optimization)
-- **Observe**: Build datasets from production traces to test against real user queries. [Learn more](/docs/observe)
-
-## Getting Started with Datasets
-
-
-
- Create datasets using SDK integration, file upload, or synthetic data generation
-
-
- Learn how to add individual records or bulk import data rows
+
+
+ Upload a file, use the SDK, or generate one from a schema
-
- Extend your dataset structure with additional data fields
+
+ How order, storage, and ownership work under the hood
-
- Test and execute prompts against your dataset entries
-
-
- Design and conduct controlled experiments to compare approaches
-
-
- Add metadata and annotations to enrich your dataset
+
+ What changes once a column is dynamic: status, edits, and reruns
-
-## Next Steps
-
-- [Understanding Datasets](/docs/dataset/concept/understanding-dataset): Deeper dive into dataset concepts, column types, and best practices
-- [Generate Synthetic Data](/docs/quickstart/generate-synthetic-data): Create realistic datasets from scratch when real data is unavailable
-- [Import from HuggingFace](/docs/cookbook/quickstart/huggingface-dataset-import): Bring existing HuggingFace datasets into Future AGI
-
diff --git a/src/pages/docs/dataset/reference/dynamic-column-methods.mdx b/src/pages/docs/dataset/reference/dynamic-column-methods.mdx
new file mode 100644
index 00000000..0ee11f7a
--- /dev/null
+++ b/src/pages/docs/dataset/reference/dynamic-column-methods.mdx
@@ -0,0 +1,119 @@
+---
+title: "Dynamic column methods"
+description: "Reference catalog of every dynamic column method, one entry per method"
+---
+
+**+ Add Columns > Dynamic Columns** opens the methods that compute a [dynamic column](/docs/dataset/concepts/static-and-dynamic-columns)'s values instead of you typing them in.
+
+## Name, Concurrency, and Status
+
+Every method's form asks for a **Name** for the resulting column, and every method except Conditional Node also asks for a **Concurrency**: how many rows to process in parallel. Retrieval's forms note that leaving Concurrency blank falls back to the platform's own system configuration. Conditional Node's form only has a **Name** and the branch list; each branch's operation carries its own Concurrency field instead, and skips its own Name field since the conditional column already has one.
+
+The column's status is `Running` while the method runs, then lands on `Completed` or `Failed`. Each cell carries its own status of `running`, `pass`, or `error`. See [Limits & Data Types](/docs/dataset/reference/limits-and-data-types) for the full list of status values.
+
+## Run Prompt
+
+Produces a value from one inference call per row, using a prompt template.
+
+| Field | Description |
+|---|---|
+| Prompt | One or more messages (system, user, assistant); reference other columns with `{{column_name}}` |
+| Model type | LLM, Text-to-Speech, Speech-to-Text, or Image Generation |
+| Model | The model to run |
+
+The resulting column records source `run_prompt`.
+
+## Retrieval
+
+Produces a value by querying a vector database index and returning matching chunks for each row.
+
+Choose a **Vector Database**: Pinecone, Qdrant, or Weaviate. These fields are shared across all three:
+
+| Field | Description |
+|---|---|
+| Column | The column whose value is sent as the query |
+| Number of chunks to fetch | How many top matches to fetch (topK) |
+| Embedding Configuration | Type (OpenAI, Hugging Face, or Sentence Transformers) and Model |
+| Key to extract | The field to pull from each retrieved match |
+| Vector Length | The dimension the embedding model outputs; must match the index's configured dimension |
+
+Each provider adds a few fields of its own, including its own API key field:
+
+| Provider | Additional fields |
+|---|---|
+| Pinecone | Pinecone API Key, Index Name, Namespace, Query Key |
+| Qdrant | Qdrant API Key, Qdrant URL, Collection Name |
+| Weaviate | Weaviate Api Key, Weaviate Cluster Url, Collection Name, Search Type (Semantic Search or Hybrid) |
+
+The resulting column records source `vector_db`.
+
+## Extract Entities
+
+Produces a value extracted from a text column, guided by a model.
+
+| Field | Description |
+|---|---|
+| Column | The column to extract from |
+| Instructions | What to extract |
+| Model | The model to run |
+
+The resulting column records source `extracted_entities`.
+
+## Extract a JSON Key
+
+Produces a value pulled out of a JSON column by key.
+
+| Field | Description |
+|---|---|
+| Column | A column of type JSON, or an API Call column whose response is JSON |
+| JSON Key | The JSONPath-style key to extract, e.g. `age` |
+
+The resulting column records source `extracted_json`.
+
+## Classification
+
+Produces a label assigned to a column's text from your set of categories.
+
+| Field | Description |
+|---|---|
+| Column | The column to classify |
+| Labels | One or more category labels |
+| Model | The model to run |
+
+The resulting column records source `classification`.
+
+## API Calls
+
+Produces a value returned by calling an external HTTP endpoint for each row.
+
+| Field | Description |
+|---|---|
+| Add API Endpoint | The endpoint to call; reference other columns with `{{column_name}}` |
+| Request Type | GET, POST, PUT, DELETE, or PATCH |
+| Params / Headers | Key-value pairs; each value is plain text, a stored secret, or a column reference |
+| Request Body | JSON body; reference other columns with `{{column_name}}` |
+| Output Type | String, Object, Array, or Number |
+
+The resulting column records source `api_call`.
+
+## Conditional Node
+
+Produces a value chosen by evaluating branches in order: the first branch whose condition is true, or the `else` branch, runs its operation, and that operation's output becomes the cell's value.
+
+| Field | Description |
+|---|---|
+| Branches | One `if` (always first), any number of `elif`, and optionally one `else` |
+| Condition | Set on every branch except `else`; reference other columns with `{{column_name}}` |
+| Select Column Type | Per branch, one of Run Prompt, Retrieval, Extract Entities, Extract JSON Key, Execute Custom Code, Classification, or API Calls, configured with that operation's own fields described on this page |
+
+The resulting column records source `conditional`.
+
+## Execute Custom Code
+
+Produces a value returned by a Python function you write, run once per row. The function must be named `main` and can read any column's value through `kwargs` (`kwargs.get("column_name")`). Execute Custom Code isn't a standalone tile under Dynamic Columns. It's only reachable as an operation inside a Conditional branch, or by editing a column that already runs Python code.
+
+| Field | Description |
+|---|---|
+| Code | The `main(**kwargs)` function to run |
+
+The resulting column records source `python_code`.
diff --git a/src/pages/docs/dataset/reference/limits-and-data-types.mdx b/src/pages/docs/dataset/reference/limits-and-data-types.mdx
new file mode 100644
index 00000000..5c5283bc
--- /dev/null
+++ b/src/pages/docs/dataset/reference/limits-and-data-types.mdx
@@ -0,0 +1,58 @@
+---
+title: "Limits & Data Types"
+description: "Quick-reference tables for Dataset limits, data types, and status values."
+---
+
+A lookup page for the numbers and enums referenced elsewhere in the Dataset docs: what each column data type stores, the exact limits on names, rows, columns, and requests, and the status values a column or a cell can be in.
+
+## Column data types
+
+| Type | Stores |
+|---|---|
+| `text` | A text value |
+| `boolean` | True or false |
+| `integer` | A whole number |
+| `float` | A decimal number |
+| `json` | A JSON object |
+| `array` | A JSON array |
+| `image` | A single image |
+| `images` | Multiple images |
+| `datetime` | A date and time value |
+| `audio` | An audio file |
+| `document` | A document file |
+| `persona` | A persona definition |
+| `others` | A value that doesn't fit any other type |
+
+## Dataset and column limits
+
+| Limit | Value | Applies to |
+|---|---|---|
+| Dataset name length | 2000 characters, unique within the organization | Every dataset |
+| Column name length | 255 characters | Every column you name |
+| Manual dataset rows | 100 | Creating a dataset manually |
+| Manual dataset columns | 100 | Creating a dataset manually |
+| Empty dataset rows | 10 | Creating an empty dataset |
+| Empty rows per request | 100 | Adding empty rows to an existing dataset |
+| Row duplication | 100 copies | Duplicating a row |
+| Bulk delete | 50 items | Deleting datasets in bulk |
+| Dataset list page size | 100 datasets | Listing datasets |
+| File upload size | 25 MB | Uploading a file to create or add to a dataset |
+| File upload formats | `.csv`, `.xls`, `.xlsx`, `.json`, `.jsonl` | Uploading a file to create or add to a dataset |
+
+
+ These are per-request and per-object constants. Plan-level quotas, such as the total rows or datasets your organization can hold, are enforced separately by the usage system and aren't part of this table.
+
+
+## Status values
+
+Column statuses and cell statuses are separate sets, and a column only reaches the ones below.
+
+| Status | Seen on | Meaning |
+|---|---|---|
+| `Running` | Column | Set when a column starts an async run, such as Run Prompt, an evaluation, or Retrieval |
+| `PartialExtracted` | Column | Set when file upload extraction succeeds for some of a column's cells and fails for others |
+| `Completed` | Column | The default status for a new column, and set when a column finishes running successfully |
+| `Failed` | Column | Set when a column's processing raises an error, for example during a data type conversion |
+| `pass` | Cell | Computed successfully (the default) |
+| `running` | Cell | Set while a cell is computing |
+| `error` | Cell | Set when a cell fails to compute |
diff --git a/src/pages/docs/dataset/troubleshooting.mdx b/src/pages/docs/dataset/troubleshooting.mdx
new file mode 100644
index 00000000..c0977632
--- /dev/null
+++ b/src/pages/docs/dataset/troubleshooting.mdx
@@ -0,0 +1,42 @@
+---
+title: "Dataset FAQ & fixes"
+description: "Common dataset questions, and fixes for the errors you hit most"
+---
+
+## In this page
+
+The questions people ask most about datasets, and the errors they run into, with a direct fix for each, in the table below. If your answer isn't here, reach out via [support](https://futureagi.com/contact-us).
+
+## Common errors and fixes
+
+| Symptom | Cause | Fix |
+|---|---|---|
+| Upload rejected before it starts | The file isn't `.csv`, `.xls`, `.xlsx`, `.json`, or `.jsonl`, or it's over 25 MB | Convert or split the file to fit; see [Limits & Data Types](/docs/dataset/reference/limits-and-data-types) for every cap |
+| "A dataset with this name already exists in your organization" when creating | Names must be unique per organization, so the clash can be with a dataset you don't have access to see | Pick a different name |
+| Upload finished but the dataset shows no rows yet | The file is still processing in the background | Wait it out; the rows land once it finishes |
+| Image, audio, or document cells fill in slowly after upload | Media cells upload in batches with retries, not all at once | No action needed, it catches up on its own |
+| A dataset built from [Observe](/docs/observe) traces fills in gradually | Spans convert into rows in chunks, not all at once | Wait it out; the row count climbs while the conversion runs |
+| A freshly added eval column sits idle for a few seconds | Evals are picked up by a poller on a short cycle rather than dispatched the instant you add the column | Give it a few seconds; it starts on its own |
+| A cell is stuck showing running | The [producer](/docs/dataset/concepts/static-and-dynamic-columns) behind it, a run prompt, an eval, or another [dynamic column method](/docs/dataset/reference/dynamic-column-methods), hasn't been picked up yet or is still executing | Wait a few seconds; if the column status shows `Failed` or `Error`, rerun the column |
+| A cell shows error | That row failed whatever produces the column | Rerun the column |
+| Synthetic generation fails | It runs as a background job and can fail partway through | Reopen it with **Configure Synthetic Data**; the drawer returns with your saved configuration, so fix whatever caused the failure and generate again |
+| A column or row you expected is gone | Most likely it was deleted; the grid can also hide columns and filter rows. Deleting a column also removes the producer behind it (a run prompt, an eval) and anything derived from it | There's no copy to restore; recreate the column or add the rows back |
+| Add Row or Add Column is greyed out | Both are disabled while the dataset is processing or synthetic generation hasn't finished; Add Column alone also stays disabled until the dataset has at least one row | Wait for processing or synthetic generation to finish, or add a row first if you're adding a column to an empty dataset |
+| Duplicate or Delete is greyed out | Permission gates both: duplicating a row needs update access, deleting one needs delete access | Check with an org admin about your dataset permission |
+
+## Keep exploring
+
+
+
+ Every way to get a dataset that exists and has data in it
+
+
+ Rename, duplicate, export, and delete once the data is in
+
+
+ Where a column's values come from, and what deleting one takes with it
+
+
+ Every row, column, and file cap in one table
+
+
diff --git a/src/pages/docs/error-feed/concepts/how-it-works.mdx b/src/pages/docs/error-feed/concepts/how-it-works.mdx
deleted file mode 100644
index dc003468..00000000
--- a/src/pages/docs/error-feed/concepts/how-it-works.mdx
+++ /dev/null
@@ -1,94 +0,0 @@
----
-title: "How Error Feed Works: Trace Analysis and Issue Grouping"
-description: "The mental model behind Error Feed: how traces become analyzed issues, how similar errors are grouped, and how findings surface in the UI."
----
-
-## About
-
-This page walks through what happens between a raw trace arriving and an issue showing up in the Feed. It's the mental model, not the internals.
-
-## The pipeline in four steps
-
-
-
- Every trace sent to a Future AGI Observe project is a candidate. Error Feed works on a sample, configurable per project. See [Sampling](/docs/error-feed/features/sampling).
-
- No extra instrumentation needed. If your agent is already instrumented with any of the [supported integrations](/docs/error-feed/#supported-integrations), it's already sending what Error Feed needs.
-
-
- For every sampled trace, Error Feed:
-
- - Reads the full span tree: inputs, outputs, tool calls, LLM responses, errors, metadata
- - Checks for failures across the [error taxonomy](/docs/error-feed/concepts/taxonomy) (five categories covering reasoning, safety, tool failures, workflow gaps, and reflection)
- - Scores the trace on four quality dimensions, 0–5: Factual Grounding, Privacy & Safety, Instruction Adherence, Optimal Plan Execution
-
- Traces that pass without issues still get scored. A score isn't a severity, it's a quality signal.
-
-
- When multiple traces fail in semantically similar ways (same error type, same part of the workflow), Error Feed groups them into a single **issue**. The cluster name describes what's going wrong, e.g. "Hallucinated entity in product lookup".
-
- The number of traces in a cluster is its **trace count**. One cluster might represent a single noisy span seen once; another might represent a systematic failure across thousands of traces.
-
- The point is to triage *problems*, not individual trace failures.
-
-
- Each issue in the list shows:
-
- - Error name and its [taxonomy category](/docs/error-feed/concepts/taxonomy)
- - [Severity](/docs/error-feed/concepts/severity-and-status) (Critical / High / Medium / Low)
- - [Status](/docs/error-feed/concepts/severity-and-status) (Unresolved, Acknowledged, Resolved, Escalating)
- - Trace count (cluster size)
- - A sparkline showing whether the issue is getting worse, improving, or stable
-
- Click an issue to open the [detail view](/docs/error-feed/features/issue-overview): description, root causes, evidence, agent flow, recommendations, and every trace in the cluster.
-
-
-
-## Two levels of analysis
-
-There are two kinds of analysis, at different cost points:
-
-**Continuous scan** runs automatically on every sampled trace. It produces the description, root cause, immediate fix, long-term recommendation, evidence snippets, and quality scores on the Overview tab. Always on.
-
-**[Deep Analysis](/docs/error-feed/features/deep-analysis)** is on-demand. It runs a more thorough investigation on the cluster's representative trace, producing more detailed pattern analysis and recommendations. Trigger it manually from the metadata panel when the continuous scan finds something worth digging into.
-
-## What "representative trace" means
-
-When a cluster has many traces, Error Feed picks one to stand in for the cluster throughout the detail view. The Overview tab's analysis, Agent Flow diagram, and Deep Analysis all run against this representative trace. The Traces tab shows every trace in the cluster, so you can jump to any specific one.
-
-## Continuous vs. sampled coverage
-
-Error Feed doesn't analyze 100% of traces by default. The sampling rate controls what fraction get analyzed. Lower rates are faster and cheaper but miss infrequent errors. 100% gives full coverage at higher cost.
-
-Important: issues are formed only from traces that were actually analyzed. At 20% sampling, five identical errors in a batch of 25 traces might show up in the cluster as one occurrence — the one that got sampled.
-
-See [Sampling](/docs/error-feed/features/sampling) for per-project configuration.
-
-
-Set sampling to 100% during development or testing. In production with high volume, 10–20% is a reasonable starting point.
-
-
-## Scores vs. errors
-
-A trace can have a low quality score with no detected error, or a detected error with otherwise decent scores. The score measures overall quality on four dimensions; error detection flags specific failure patterns from the taxonomy. Both show up in the issue detail: scores in the metadata panel's Evaluations section, errors on the Overview tab.
-
-If an issue has a low Factual Grounding score but no Hallucinated Content error was flagged, that's still worth a look — the classifier missed something the score caught.
-
-***
-
-## Next steps
-
-
-
- What error types Error Feed detects and how they're categorized.
-
-
- How the four quality dimensions are defined and how to interpret scores.
-
-
- The issue list page — filters, columns, and how to navigate it.
-
-
- Configure what percentage of traces Error Feed analyzes.
-
-
diff --git a/src/pages/docs/error-feed/concepts/scoring.mdx b/src/pages/docs/error-feed/concepts/scoring.mdx
deleted file mode 100644
index 7f841c3d..00000000
--- a/src/pages/docs/error-feed/concepts/scoring.mdx
+++ /dev/null
@@ -1,90 +0,0 @@
----
-title: "Error Feed Quality Scoring: Four Trace Metrics Explained"
-description: "The four quality metrics Error Feed uses to score every analyzed trace, what each one measures, how scores are assigned, and how to interpret them."
----
-
-## About
-
-Every analyzed trace gets scored on four quality dimensions, 0 to 5, where 5 is best. Scores show up in the metadata panel's Evaluations section and in the Trends tab's Score Trends chart.
-
-Scoring is separate from error detection. A trace can score badly on one dimension without triggering a classified error, and a trace with a detected error can still score fine on unrelated dimensions. Both are useful: scores give you a continuous quality gradient, error detection gives you discrete failure labels.
-
-
-
-## The four dimensions
-
-### Factual Grounding
-
-How well the agent's output is anchored in verifiable evidence: the retrieved context, provided documents, or facts the agent had access to when it responded.
-
-A low score means the output makes claims the input data doesn't support. This is the main signal for hallucination risk. An agent confidently answering with information it couldn't have derived from its context will score low here.
-
-Common causes:
-- Retrieving the wrong chunks (or failing to retrieve at all) and answering anyway
-- Summarizing beyond what the source actually says
-- Inventing specific details like names, numbers, or dates
-
-### Privacy & Safety
-
-How well the agent follows safety and security practices: PII protection, credential hygiene, safe advice, output fairness.
-
-A low score means the output may expose personal data, leak credentials, give advice that could cause harm, or contain biased content. This matters most for agents that handle user data, hit external services, or operate in sensitive domains.
-
-Common causes:
-- Including user names, emails, phone numbers, or IDs in outputs that shouldn't have them
-- Echoing API keys or tokens from tool responses back into text
-- Generating advice with material risk attached (medical, legal, financial)
-- Producing content that stereotypes groups
-
-### Instruction Adherence
-
-How faithfully the agent follows the instructions it's been given: system prompt, user instructions, formatting constraints, tone guidelines, task-specific rules.
-
-A low score means the agent did something it was told not to do, skipped something it was told to do, or produced output in the wrong format. This catches prompt compliance failures that don't look like "errors" in the traditional sense.
-
-Common causes:
-- Responding in prose when structured JSON was required
-- Ignoring a "respond only in English" constraint
-- Answering a question the system prompt explicitly says to deflect
-- Skipping required fields in a structured output schema
-
-### Optimal Plan Execution
-
-The quality of the agent's decision-making: whether it picked the right tools, in the right order, with the right parameters, and structured its multi-step workflow logically.
-
-A low score means the plan was inefficient, wrong, or incomplete. The agent may have used the wrong tool, called the same one repeatedly without a clear reason, executed steps out of order, or abandoned the task before finishing it.
-
-Common causes of low scores:
-- Selecting a less-capable tool when a more appropriate one was available
-- Calling a tool with incorrect or missing parameters
-- Executing steps in an illogical order
-- Abandoning a multi-step workflow before reaching a conclusion
-
-## How scores appear in the UI
-
-The metadata panel on the right of any issue detail page has an Evaluations section. Each entry is an eval run against the representative trace. LLM-judge evals show a score bar with a percentage; pass/fail evals show a pass or fail verdict.
-
-The [Trends tab](/docs/error-feed/features/trends) shows score trends over time so you can see whether quality is improving or degrading across the cluster.
-
-
-Scores in the metadata panel reflect whichever trace is selected in the Traces tab. Switch traces and the scores update.
-
-
-## Scores vs. detected errors
-
-The scoring dimensions intentionally overlap with the [error taxonomy](/docs/error-feed/concepts/taxonomy). A Factual Grounding score of 1/5 lines up with a Hallucinated Content error. A Privacy & Safety score of 2/5 might pair with a PII Leak.
-
-They're not the same thing though. The score is continuous, 0 to 5. The error classification is a discrete label saying "this specific failure pattern was detected." Both can point to the same problem; using them together gives you a clearer picture.
-
-A cluster with consistently low scores on all four dimensions is a fundamentally broken workflow, not a narrow edge case. When only one dimension is low, the problem is more targeted.
-
-## Next Steps
-
-
-
- How issues are classified and how to move them through triage.
-
-
- Score trends over time to see if quality is improving.
-
-
diff --git a/src/pages/docs/error-feed/concepts/severity-and-status.mdx b/src/pages/docs/error-feed/concepts/severity-and-status.mdx
index 6baeea0e..c10565d1 100644
--- a/src/pages/docs/error-feed/concepts/severity-and-status.mdx
+++ b/src/pages/docs/error-feed/concepts/severity-and-status.mdx
@@ -1,80 +1,59 @@
---
-title: "Error Feed Issue Severity and Triage Status"
-description: "Error Feed severity levels classify how critical each issue is, and status labels track issues through triage from Open to Resolved."
+title: "Severity & Status"
+description: "How an issue's triage state and its impact level move independently"
---
-## About
+## Two axes, one issue
-Every issue has two independent labels: **severity** and **status**. Severity is how bad the problem is. Status is where it sits in your triage workflow. Different purposes, updated independently.
+Every [issue](/docs/error-feed/concepts/understanding-error-feed) in the [feed](/docs/error-feed/concepts/understanding-error-feed) carries two labels that change independently of each other. **Status** is the [triage](/docs/error-feed/guides/triage-issues) axis: where the issue sits in your team's workflow. **Severity** is the impact axis: how bad the problem is. A newly created issue starts at status Escalating and severity Medium.
-## Severity
+Nothing about either axis moves on its own. Every change is a person picking a new value, and any value can jump straight to any other, in either direction, at any time.
-Severity reflects impact and urgency. It's assigned automatically based on the error type and quality scores, but you can override it manually on any issue.
+## Status: the triage axis
-| Severity | What it means |
-|----------|---------------|
-| **Critical** | High-confidence, high-impact failure. Likely affecting users now. Examples: safety violations, authentication failures, data exposure, complete task abandonment. |
-| **High** | Significant problem with clear user impact. Examples: consistent hallucination, systematic tool misuse, repeated workflow failures. |
-| **Medium** | Notable quality degradation, but not catastrophic. Examples: instruction adherence drift, suboptimal tool choices, inconsistent formatting. |
-| **Low** | Minor issue or edge case. Low frequency or low impact. Examples: slight verbosity, mild instruction drift on an uncommon input. |
+Status has exactly four values.
-Severity shows up as a colored badge in the issue list and the issue header. Change it any time from the severity dropdown in the [metadata panel](/docs/error-feed/features/metadata-panel).
+| Status | What it means |
+|---|---|
+| Escalating | The default for a newly created issue. Nothing has been decided about it yet |
+| For review | Flagged for a closer look before someone decides what to do |
+| Acknowledged | Confirmed as real |
+| Resolved | The underlying problem has been fixed |
-
-Use severity as a prioritization signal. Start every triage session with Critical and High. Low-severity issues are worth tracking but rarely need immediate action.
-
+These are the same four values you'll see spelled `escalating`, `for_review`, `acknowledged`, and `resolved` in the data and filters, just written out for reading here.
-## Status
+## Severity: the impact axis
-Status tracks where the issue is in your workflow. Severity describes the problem; status describes what your team has decided to do about it.
+Severity also has exactly four values, describing how bad the problem is: critical, high, medium, and low, with medium as the default for a newly created issue. That's a different question from the quality [scores](/docs/error-feed/concepts/trace-error-analysis) a trace receives.
-| Status | Meaning |
-|--------|---------|
-| **Unresolved** | The default for any new issue. No one has looked at it yet, or it's been reviewed but not acted on. |
-| **Acknowledged** | Someone has seen this issue and confirmed it's real. Work may or may not be underway. Use this to signal "we know about it" to the rest of the team. |
-| **Resolved** | The underlying problem has been fixed. Error Feed will continue monitoring, and if the same pattern resurfaces it will create a new issue. |
-| **Escalating** | The issue is actively getting worse — increasing frequency, expanding impact, or a fix hasn't held. Use this to flag urgency. |
+
+Severity is stored as `priority` on the issue: critical maps to `urgent`, high to `high`, medium to `medium`, and low to `low`.
+
-A **For Review** status is also available from the dropdown for issues that need a closer look before someone decides what to do.
+Severity is a label your team sets, then filters and sorts by. See [Issue fields & filters](/docs/error-feed/reference/issue-fields) for the full value table and the feed's other filterable fields.
-## Changing status
+## How the two axes relate
-Three ways:
+|"triage axis"| STATUS["Status · escalating, for review, acknowledged, resolved"]
+ ISSUE -->|"impact axis"| SEVERITY["Severity · critical, high, medium, low"]
+`} />
-1. **Header action buttons**: the issue detail header has Resolve, Acknowledge, and Ignore issue buttons for one-click updates.
-2. **Status dropdown in the metadata panel**: click the current status chip in the Triage section of the right sidebar to switch to any state.
-3. **Triage workflow**: see [Triage Workflow](/docs/error-feed/features/triage-workflow) for the full picture, including assigning issues.
+Because the two axes are independent, an issue can sit in any combination of the two. A critical issue can still be sitting at escalating: its impact is as bad as it gets, but no one has picked it up to move it forward yet.
-
-Resolving an issue doesn't suppress future detection. If the same error pattern shows up again after a fix, it creates a new issue. That's deliberate: regressions should be visible.
-
+## Why it matters
-## The "first seen" marker
+Keeping status and severity separate means status can say nothing about how bad an issue is, and severity can say nothing about where it stands in triage. Collapsing them into one field would lose that distinction, and with it the ability to tell "critical but untouched" apart from "already being worked."
-Recently detected issues show a **first seen** indicator in the feed. This makes it easy to spot new issues without opening every one.
-
-The metadata panel's Timeline section has the exact first-seen, last-seen, and age (in days) for every issue.
-
-## How severity and status interact
-
-They're independent. An issue can be Critical and Acknowledged (you know it's bad, you're working on it), or Low and Escalating (started small but keeps coming up). Set each accurately rather than using one as a proxy for the other.
-
-The feed list is sortable and filterable by both. Typical workflow:
-
-1. Filter to **Unresolved + Critical** to find the most urgent uninvestigated issues
-2. Acknowledge issues you've reviewed, assign them to the right person
-3. Mark Resolved once the fix is deployed and verified
-4. Watch for the same cluster reappearing — if it does, the fix didn't hold
-
-See [Triage Workflow](/docs/error-feed/features/triage-workflow) for step-by-step guidance on working through a batch of issues.
-
-## Next Steps
+## Keep exploring
-
- Step-by-step guidance for working through a backlog.
+
+ The per-trace Scores accordion, a separate view that doesn't drive an issue's status or severity
-
- Filter issues by severity and status.
+
+ Change an issue's status or severity, then filter and act on the feed
diff --git a/src/pages/docs/error-feed/concepts/taxonomy.mdx b/src/pages/docs/error-feed/concepts/taxonomy.mdx
deleted file mode 100644
index 1d29aefb..00000000
--- a/src/pages/docs/error-feed/concepts/taxonomy.mdx
+++ /dev/null
@@ -1,125 +0,0 @@
----
-title: "Error Feed Taxonomy: Five AI Agent Error Categories"
-description: "Reference for the five categories of errors Error Feed detects in AI agent traces, with every subcategory and error type defined."
----
-
-## About
-
-Error Feed classifies every detected failure into one of five top-level categories. Each one covers a distinct class of agent failure: bad reasoning, broken tools, unsafe output, and so on. Knowing the taxonomy helps you figure out where to look when an issue lands in the feed.
-
-
-
-The five categories:
-
-- **Thinking & Response Issues**: failures in reasoning, factual grounding, and output quality
-- **Safety & Security Risks**: outputs or behaviors that could cause harm, expose data, or break security practices
-- **Tool & System Failures**: errors from broken tools, APIs, or execution environments
-- **Workflow & Task Gaps**: breakdowns in multi-step orchestration, memory, and retrieval
-- **Reflection Gaps**: failures to reason through problems or self-correct
-
-***
-
-## Thinking & Response Issues
-
-Mistakes in understanding, reasoning, factual grounding, or output formatting.
-
-| Subcategory | Error Type | Description |
-|-------------|------------|-------------|
-| **Hallucination Errors** | Hallucinated Content | Output includes information that is invented or not supported by input data. |
-| | Ungrounded Summary | Summary includes claims not found in the retrieved chunks or original context. |
-| **Information Processing** | Poor Chunk Match | Retrieved irrelevant or unrelated context. |
-| | Wrong Chunk Used | Response based on wrong part of retrieved content. |
-| | Tool Output Misinterpretation | Misread or misunderstood the output returned by a tool or API. |
-| **Decision Errors** | Wrong Intent | Misunderstood the core user goal or instruction. |
-| | Tool Misuse | Used a tool incorrectly or in the wrong context. |
-| | Wrong Tool Chosen | Selected an inappropriate tool for the task. |
-| | Invalid Tool Params | Passed malformed, missing, or incorrect parameters to a tool. |
-| | Missed Detail | Skipped a key part of the user prompt or prior context. |
-| **Format & Instruction** | Bad Format | Output is not valid JSON, CSV, or code. |
-| | Instruction Adherence | Didn't follow instruction or style. |
-
-***
-
-## Safety & Security Risks
-
-Any output or behavior that may cause harm, leak personal data, or violate security best practices.
-
-| Subcategory | Error Type | Description |
-|-------------|------------|-------------|
-| **Ethical Violations** | Unsafe Advice | Could lead to harm if followed. |
-| | PII Leak | Sensitive personal info exposed in output. |
-| | Biased Output | Stereotyped, unfair, or discriminatory content. |
-| **Security Failures** | Token Exposure | Secrets, API keys, or auth tokens were exposed in output or logs. |
-| | Insecure API Usage | Used HTTP instead of HTTPS, skipped auth headers, or lacked rate limits. |
-
-***
-
-## Tool & System Failures
-
-Errors due to tool, API, environment, or runtime failures.
-
-| Subcategory | Error Type | Description |
-|-------------|------------|-------------|
-| **Setup Errors** | Tool Missing | Tool not registered or available. |
-| | Tool Misconfigured | Tool or API setup is incorrect (e.g., bad schema, invalid registration). |
-| | Env Incomplete | Missing tokens, secrets, or setup environment variables. |
-| **Tool/API Failures** | Rate Limit | Too many requests hit the limit. |
-| | Auth Fail | Authentication to tool or service failed. |
-| | Server Crash | Tool/API returned internal error. |
-| | Resource Not Found | Requested endpoint or resource does not exist or is not reachable. |
-| **Runtime Limits** | Out of Memory | RAM or resource limit breached. |
-| | Timeout | Execution took too long and was halted. |
-
-***
-
-## Workflow & Task Gaps
-
-Breakdowns in multi-step task execution, orchestration, or memory.
-
-| Subcategory | Error Type | Description |
-|-------------|------------|-------------|
-| **Context Loss** | Dropped Context | Missed relevant past messages or data. |
-| | Overuse | Unnecessary context/tools invoked. |
-| **Retrieval Errors** | Poor Chunk Match | Retrieved irrelevant or unrelated context. |
-| | Wrong Chunk Used | Response based on wrong part of retrieved content. |
-| | No Retrieval | Failed to run retrieval when needed. |
-| **Task Flow Issues** | Goal Drift | Strayed from intended objective. |
-| | Step Disorder | Steps executed out of logical order. |
-| | Redundant Steps | Repeated same tool or action unnecessarily. |
-| | Task Orchestration Failure | Agent failed to plan or interleave actions properly across tools or steps. |
-| **Trace Completion** | Incomplete Task | No final result or closure. |
-
-***
-
-## Reflection Gaps
-
-Agent failed to engage in introspective reasoning or revise steps appropriately.
-
-| Error Type | Description |
-|------------|-------------|
-| Missing CoT | No intermediate thinking steps (Chain of Thought) were used to justify actions. |
-| Missing ReAct Planning | Agent failed to interleave reasoning with action; took action without planning. |
-| Lack of Self-Correction | Agent didn't revise response or plan after detecting error or contradiction. |
-
-***
-
-## How taxonomy categories appear in the UI
-
-On the [Overview tab](/docs/error-feed/features/issue-overview), each detected error shows its taxonomy type as a chip. The Description section says what went wrong in this trace; Root Cause says why; Evidence quotes the relevant spans directly.
-
-On the [Feed list](/docs/error-feed/features/the-feed), issues are tagged with their primary error type so you can filter by category when you're hunting a specific class of failure.
-
-
-A single trace can trigger errors in multiple categories. A tool failure that causes the agent to hallucinate a fallback answer will register as both a Tool & System Failure and a Thinking & Response Issue.
-
-
-## Next Steps
-
-
-
- Where taxonomy categories appear in the issue detail UI.
-
-
- How quality scores complement error detection.
-
-
diff --git a/src/pages/docs/error-feed/concepts/trace-error-analysis.mdx b/src/pages/docs/error-feed/concepts/trace-error-analysis.mdx
new file mode 100644
index 00000000..c621af39
--- /dev/null
+++ b/src/pages/docs/error-feed/concepts/trace-error-analysis.mdx
@@ -0,0 +1,50 @@
+---
+title: "Trace error analysis"
+description: "The four per-trace quality scores and why they stay out of the feed"
+---
+
+## What Error Analysis is
+
+Open a [trace](/docs/observe/concepts/traces) that has one and you'll find **Error Analysis** in the **Scores** accordion: four quality dimensions, each scored for that one trace. It's a separate, per-trace view, not a property of an [issue](/docs/error-feed/concepts/understanding-error-feed) or a cluster.
+
+## The four dimensions
+
+- **Factual Grounding**: whether the response holds up against the evidence and context the agent actually had
+- **Privacy And Safety**: whether the response handles sensitive data and follows safe practices
+- **Instruction Adherence**: whether the response follows the instructions the agent was given
+- **Optimal Plan Execution**: whether the agent's sequence of decisions and tool calls was the right one for the task
+
+
+The UI title-cases the raw dimension name, so what you'd write as "Privacy & Safety" renders as Privacy And Safety in the product.
+
+
+## Where you see it
+
+Open the [trace detail drawer](/docs/observe/guides/explore-dashboard#open-a-trace) and its **Scores** accordion shows one chip per dimension: the label and the score out of 5, for example "Factual Grounding 4/5".
+
+## When to check the scores
+
+Read the scores when you're already looking at a specific trace, for example while working through an issue in the [Investigate an issue](/docs/error-feed/guides/investigate-an-issue) guide, and want a read on that trace beyond the issue's category. Treat a dimension scoring lower than the others as a pointer to look closer at the plan, the tool calls, or the response, not a verdict on its own.
+
+## A different pipeline from the Error Feed scanner
+
+|Error Feed scanner| FD["Finding → issue in the feed"]
+ TR -->|Error Analysis| SC["Four dimension scores → Scores accordion"]`} />
+
+Error Analysis and the [Error Feed](/docs/error-feed) scanner are two independent reads on the same trace. The feed reads traces, groups the problems it finds into issues, and tells you where to fix them; the Scores accordion tells you how one specific trace performed on these four dimensions.
+
+These four scores don't feed the [error taxonomy](/docs/error-feed/reference/error-taxonomy): a low score doesn't create an issue, and it isn't a value you can filter the feed by. Checking both means opening the trace and reading the accordion yourself.
+
+## Keep exploring
+
+
+
+ Open an issue and work through the evidence, trace by trace
+
+
+ The fixed set of groups, categories, and fix layers a scan can assign
+
+
diff --git a/src/pages/docs/error-feed/concepts/understanding-error-feed.mdx b/src/pages/docs/error-feed/concepts/understanding-error-feed.mdx
new file mode 100644
index 00000000..3119e1ad
--- /dev/null
+++ b/src/pages/docs/error-feed/concepts/understanding-error-feed.mdx
@@ -0,0 +1,53 @@
+---
+title: "Understanding Error Feed"
+description: "How a scan or eval failure becomes one issue, and what that issue actually owns"
+---
+
+## What an issue is
+
+[Error Feed](/docs/error-feed) reads your [traces](/docs/observe/concepts/traces) and turns the problems it finds in them into issues you triage.
+
+Nothing marks a trace as failing beforehand. Error Feed samples traces at whatever rate the project is set to, reads each sampled trace in full, and decides for itself whether something went wrong in it. A trace is "failing" only in the sense that a scan found at least one problem in it, which is why raising the sampling rate surfaces more issues: it isn't finding more failures, it's reading more traces.
+
+An **issue** isn't one such trace. It's the one problem behind many of them. When ten traces go wrong the same way, the feed doesn't hand you ten rows to read one by one, it hands you the single issue they all point at, and that issue is what you work.
+
+## What an issue carries
+
+Every issue in the feed carries the same things, and each one is there to help you decide what to do about it:
+
+- **A title**, naming the problem in a line
+- **A category and a group**, the two labels that place the problem in the [error taxonomy](/docs/error-feed/reference/error-taxonomy)
+- **A fix layer**, the part of your system the fix belongs in
+- **A severity and a status**, the two independent axes covered in [Severity & Status](/docs/error-feed/concepts/severity-and-status)
+- **An assignee**, once someone picks it up
+- **How often and how widely it happened**: the number of times it fired, the number of [traces](/docs/observe/concepts/traces) affected, and the number of [users](/docs/observe/concepts/users) behind those traces
+- **When it started and when it last happened**
+- **The evidence behind it**: the traces, [spans](/docs/observe/concepts/spans), and [sessions](/docs/observe/concepts/sessions) that contributed, so you can open the exact span rather than hunting through a trace
+
+Issues also come from a failing [eval](/docs/evaluation), not only from a scan. Those group by the eval that failed, and the feed shows the eval's name where a scan issue shows its group.
+
+For example, ten traces that all call the wrong tool for a refund lookup surface as one cluster: title "Wrong tool selected for refund lookup", group Tool Failures, fix layer Tools, 34 total events across 12 unique traces and 9 users affected, first seen 09:14 and last seen 14:02.
+
+## Fix layers: where the fix belongs
+
+A fix layer is one of Prompt, Tools, Orchestration, or Guardrails. It's the product's actual answer to "so what do I do about this": it names where in your system the fix belongs, not just what went wrong. Scanner clusters always carry one, taken straight from the finding. Eval clusters carry one where it can be determined: a best-effort step tries to infer it, and it's left unset when that step can't. The error taxonomy is the reference for which specific error types map to which fix layer.
+
+Fix layer rides on the cluster itself, so it's also a live filter in the feed, letting you work through everything that needs a prompt change before you touch anything that needs an orchestration change.
+
+## Why it matters
+
+Working one issue instead of a thousand traces is what makes the feed usable at scale. A single misbehaving tool can fail on every call for an hour and put a problem in thousands of traces; read them one at a time and you're reading the same failure a thousand times, work the issue and you fix it once. And because the fix layer sits on the issue itself, the feed tells you what to change before you've opened a single trace inside it.
+
+## Keep exploring
+
+
+
+ The two independent axes every issue carries, and how they change
+
+
+ A separate per-trace scoring pipeline, and why it doesn't drive the feed
+
+
+ Open an issue and work through the evidence behind it
+
+
diff --git a/src/pages/docs/error-feed/features/deep-analysis.mdx b/src/pages/docs/error-feed/features/deep-analysis.mdx
deleted file mode 100644
index c36add3f..00000000
--- a/src/pages/docs/error-feed/features/deep-analysis.mdx
+++ /dev/null
@@ -1,96 +0,0 @@
----
-title: "Error Feed Deep Analysis: On-Demand Trace Investigation"
-description: "On-demand investigation that runs deeper root cause analysis on an Error Feed issue's trace and produces richer findings than the continuous scan."
----
-
-## About
-
-Every issue in Error Feed comes with analysis generated automatically by the continuous scan: description, root cause, evidence, recommendations. For most issues that's enough to understand and act on the problem.
-
-Deep Analysis is an additional, on-demand investigation you trigger manually. It runs a more thorough analysis of the issue's representative trace and produces richer findings, especially around root cause precision and the Recommendations & Fixes section.
-
-## When to use
-
-Use Deep Analysis when:
-
-- The continuous scan's findings feel incomplete or you want more specificity about root cause
-- The issue is high-severity and you want confidence in the diagnosis before investing in a fix
-- The Overview tab's Probable Root Cause or Recommendations sections are sparse
-- You're doing a post-incident review and need detailed evidence for a write-up
-
-You don't need to run it on every issue. Save it for the ones where the standard analysis leaves open questions.
-
-## How to run it
-
-Deep Analysis is triggered from the **Deep Analysis section** in the [metadata panel](/docs/error-feed/features/metadata-panel) on the right side of the issue detail page.
-
-
-
- Navigate to any issue from the [Feed list](/docs/error-feed/features/the-feed).
-
-
- In the right sidebar, scroll to the **Deep Analysis** section. If no analysis has been run yet, you'll see a **Run Deep Analysis** button.
-
-
- Click the button. A toast confirms the analysis has started, and you'll see a progress indicator: "Running analysis…"
-
- Takes about a minute. You can navigate away; analysis continues in the background.
-
-
- When it finishes, the metadata panel shows "Analysis complete." Go back to the **Overview tab** to see the updated findings. Probable Root Cause and Recommendations & Fixes will be populated with more detailed content.
-
-
-
-
-
-## What it produces
-
-Results appear in the [Overview tab](/docs/error-feed/features/issue-overview) under **Probable Root Cause** and **Recommendations & Fixes**, with more detail than the continuous scan produces.
-
-Specifically, Deep Analysis tends to produce:
-
-- More specific identification of the failing component or step
-- More targeted recommendations that reference the actual tool, prompt structure, or workflow pattern involved
-- Richer pattern analysis when the cluster has multiple traces with consistent failure characteristics
-
-## Re-running analysis
-
-Once Deep Analysis completes, the metadata panel shows a **Re-run** button alongside "Analysis complete." Use Re-run when:
-
-- You've made changes to your agent and want fresh analysis to confirm whether the root cause has shifted
-- The cluster has grown significantly since the last run and you want updated findings
-
-Re-running discards the previous result and generates new findings from scratch against the current representative trace.
-
-## Analysis states
-
-| State | What you see | What to do |
-|-------|--------------|------------|
-| Idle | "Run Deep Analysis" button | Click to start |
-| Running | Progress spinner + "Running analysis…" | Wait or navigate away |
-| Complete | "Analysis complete" + Re-run option | Check Overview tab for results |
-| Failed | "Retry Deep Analysis" button | Click to retry |
-
-
-If no trace is selected or the cluster has zero analyzed traces, the Deep Analysis button is disabled. It re-enables as soon as at least one trace enters the cluster.
-
-
-## Relationship to continuous scan
-
-Continuous scan runs automatically on every sampled trace. Deep Analysis runs once, on demand, against the representative trace. They're complementary:
-
-- Continuous scan gives you breadth: every trace gets analyzed, issues surface automatically
-- Deep Analysis gives you depth: one trace gets a thorough investigation when you need it
-
-Neither replaces the other. Typical pattern: continuous scan surfaces the issue, Deep Analysis helps you understand it well enough to fix it confidently.
-
-## Next Steps
-
-
-
- How Deep Analysis output appears in the Overview tab.
-
-
- When to escalate from continuous scan to deep analysis.
-
-
diff --git a/src/pages/docs/error-feed/features/issue-overview.mdx b/src/pages/docs/error-feed/features/issue-overview.mdx
deleted file mode 100644
index f47ee9c0..00000000
--- a/src/pages/docs/error-feed/features/issue-overview.mdx
+++ /dev/null
@@ -1,107 +0,0 @@
----
-title: "Error Feed Issue Overview: Header to Recommendations"
-description: "A walkthrough of the Overview tab on an issue detail page: every section explained, from the header to the AI-generated recommendations."
----
-
-## About
-
-The issue detail page is where you understand a problem well enough to fix it. It opens when you click any issue in the [Feed list](/docs/error-feed/features/the-feed). Three zones: a header, a tab area, and a metadata panel on the right.
-
-This page covers the **Overview tab**, which is the default view. For the others see [Traces](/docs/error-feed/features/traces), [State Graph](/docs/error-feed/features/state-graph), and [Trends](/docs/error-feed/features/trends). For the right sidebar see [Metadata Panel](/docs/error-feed/features/metadata-panel).
-
-
-
-## The header
-
-The header stays visible regardless of which tab you're on.
-
-
-
-**Breadcrumb**: "Error Feed" links back to the list. The chip next to it shows the error type (e.g. "Hallucination", "Tool Failure").
-
-**Error title**: the cluster name, describing what's going wrong.
-
-**Status chips row** (left to right): error type dot, [status chip](/docs/error-feed/concepts/severity-and-status), [severity badge](/docs/error-feed/concepts/severity-and-status), trace count chip.
-
-**Action buttons** (top right):
-- **Copy cluster ID**: copies the cluster identifier, handy for tickets or Slack
-- **Share**: copies a direct link to this issue
-- **Resolve**: marks the issue resolved in one click
-- **Acknowledge**: marks the issue acknowledged
-- **Ignore issue**: suppresses the issue from the default view
-
-
-The Resolve and Acknowledge buttons in the header are shortcuts. The full triage workflow (assigning to a team member, changing severity) lives in the metadata panel. See [Triage Workflow](/docs/error-feed/features/triage-workflow).
-
-
-## Overview tab layout
-
-Two columns. The left lists every trace in the cluster (the Traces affected panel) along with a small events-and-users chart. The right shows analysis for the selected trace.
-
-Click any trace on the left to focus the right column on it.
-
-## Always-visible sections
-
-These sections come from the continuous scan that runs on every sampled trace. They show up the moment you open the issue, no extra action required.
-
-### Selected trace header
-
-A compact bar at the top of the right column with the selected trace's ID, latency, cost, and total tokens. Treat it as the breadcrumb for which trace's analysis you're currently looking at.
-
-### Pattern Summary
-
-A summary of patterns across the whole cluster, not just the selected trace. The card has a one-line takeaway plus four headline metrics (e.g. "68% of errors involve retrieval step", "3.4× vs. baseline error rate", "12s median time-to-fail", "GPT-4o top-affected model").
-
-Most useful when the cluster has a large trace count, where individual trace analysis won't surface systemic patterns that are obvious at scale.
-
-### Agent Flow
-
-A visual diagram of the steps the agent took (LLM calls, tool invocations, sub-agent interactions) and where the failure happened.
-
-The diagram makes it obvious whether the error is in step one or step five, whether it's in the LLM response or a downstream tool call, and whether there's a clear bifurcation point between traces that succeed and traces that fail.
-
-
-
-### Trace Evidence
-
-A side-by-side view of a failing trace and a working trace with the differences highlighted inline. Two tabs: **Failing Trace** (default) and **Working Trace**.
-
-Each reel walks through user input, retrieved context, model output, and eval verdict, quoting the actual content with deltas color-coded so you can spot where the failing run diverged. This is the section that shows you concretely *what* went wrong.
-
-## After running Deep Analysis
-
-The continuous scan is fast and cheap, but it stops at "what happened." For *why* it happened (and what to do about it), trigger **Deep Analysis** from the metadata panel on the right. Takes about a minute.
-
-When it finishes, two new sections appear at the bottom of the Overview tab:
-
-### Probable Root Cause
-
-A ranked list of causes (usually two to four) explaining *why* the cluster is failing. Each cause has a short title and a longer explanation. Ordered by how strongly the analysis supports them, so the top one is the best candidate to investigate first.
-
-### Recommendations & Fixes
-
-A ranked list of suggested fixes, each with a priority chip (High / Medium / Low). Click any recommendation to expand it:
-
-- **Description**: what the recommendation is, in plain language
-- **Immediate Fix**: the minimal change to apply right now (often a one-liner you can paste into a prompt or config)
-- **Insights**: why this fix works and how it relates to the root cause
-- **Evidence**: the trace data or pattern that supports the recommendation
-
-Recommendations link back to the Probable Root Causes that motivated them, so you can see which fix addresses which cause.
-
-
-
-
-If Probable Root Cause and Recommendations & Fixes aren't on the Overview tab, Deep Analysis hasn't run for this issue yet. Click **Run Deep Analysis** in the metadata panel to trigger it. See [Deep Analysis](/docs/error-feed/features/deep-analysis).
-
-
-## Next Steps
-
-
-
- View every trace in the cluster.
-
-
- Trigger a deeper investigation on this issue.
-
-
diff --git a/src/pages/docs/error-feed/features/linear-integration.mdx b/src/pages/docs/error-feed/features/linear-integration.mdx
deleted file mode 100644
index e7c5c32c..00000000
--- a/src/pages/docs/error-feed/features/linear-integration.mdx
+++ /dev/null
@@ -1,75 +0,0 @@
----
-title: "Creating Linear Tickets from Error Feed Issues"
-description: "Create Linear tickets directly from Error Feed issues to link your AI error monitoring to your engineering workflow and track fixes."
----
-
-## About
-
-Error Feed integrates with Linear so you can turn a detected issue into a tracked engineering task without leaving the platform. The integration is action-only: it creates tickets from Error Feed issues. Linear issues don't sync back to Error Feed.
-
-## Prerequisites
-
-You need a Linear account and a Future AGI account with the Linear integration connected. If you haven't connected Linear yet:
-
-
-
- Navigate to the Future AGI dashboard settings and open the Integrations page.
-
-
- Find the Linear integration and follow the OAuth flow to authorize the connection. You'll need Linear admin or member access.
-
-
-
-Once connected, the Linear row in the metadata panel of every Error Feed issue will show "Connected."
-
-## Creating a ticket
-
-
-
- Navigate to any issue in the [Error Feed list](/docs/error-feed/features/the-feed) and open it.
-
-
- Scroll to the bottom of the [metadata panel](/docs/error-feed/features/metadata-panel) on the right side. The Integrations section shows the Linear row with a **Create issue** button.
-
-
- A dialog opens with your Linear teams. Pick the team you want the ticket in.
-
-
- The ticket is created in Linear and immediately linked to the Error Feed issue. The metadata panel updates to show the Linear issue ID (e.g. "ENG-1234") with a **View ENG-1234** button that opens the ticket.
-
-
-
-## What's in the ticket
-
-The Linear ticket is pre-populated with context from the Error Feed issue:
-
-- Cluster name as the ticket title
-- Error description and root cause from the Overview tab as the ticket body
-- A link back to the Error Feed issue detail page
-
-Your engineering team has everything they need to understand the problem without cross-referencing Future AGI separately.
-
-## Viewing a linked ticket
-
-Once a ticket exists, the Linear row in the metadata panel shows the issue ID and a "View [issue ID]" link. Click to open the ticket in Linear.
-
-If the issue has already been linked, clicking "Create issue" again opens the existing ticket rather than creating a duplicate.
-
-## Disconnected state
-
-If Linear isn't connected, the row shows "Not connected" with a **Connect** button. Click to go to Settings → Integrations and set up the connection.
-
-
-The Linear integration pairs well with the [triage workflow](/docs/error-feed/features/triage-workflow). A typical pattern: review the issue in Error Feed, acknowledge it, create a Linear ticket, assign the ticket to the engineer who owns that component.
-
-
-## Next Steps
-
-
-
- How Linear tickets fit into the triage process.
-
-
- The right sidebar where the Linear integration lives.
-
-
diff --git a/src/pages/docs/error-feed/features/metadata-panel.mdx b/src/pages/docs/error-feed/features/metadata-panel.mdx
deleted file mode 100644
index 599a847a..00000000
--- a/src/pages/docs/error-feed/features/metadata-panel.mdx
+++ /dev/null
@@ -1,127 +0,0 @@
----
-title: "Error Feed Metadata Panel: Triage, Stats, and Integrations"
-description: "The right-side metadata panel on an Error Feed issue page covers triage controls, cluster stats, timeline, evaluations, and co-occurring issues."
----
-
-## About
-
-The metadata panel runs along the right side of every issue detail page and stays visible regardless of tab. It's where the operational info about an issue lives: status, assignee, cluster stats, timeline, AI metadata, evaluations, linked issues, and integrations.
-
-
-
-## Triage
-
-The Triage section handles status, severity, and assignee.
-
-**Status**: click the status chip for a dropdown with all states: Escalating, Acknowledged, For Review, Resolved. The chip is color-coded by status. See [Severity and Status](/docs/error-feed/concepts/severity-and-status).
-
-**Severity**: click the severity chip to change it. Options: Critical, High, Medium, Low. Override the auto-assigned severity whenever you have better context about actual user impact.
-
-**Assignee**: click Assign to assign the issue to a team member. The dropdown lists everyone in your org. Click the assigned name to reassign or unassign.
-
-Changes take effect immediately and show up in the Feed list view.
-
-## Cluster
-
-At-a-glance stats about the issue's scope:
-
-| Field | What it shows |
-|-------|---------------|
-| **Traces** | Total number of traces grouped in this cluster |
-| **Users affected** | Distinct users whose traces appear in the cluster |
-| **Sessions** | Number of distinct sessions represented |
-| **Cluster ID** | The unique identifier for this cluster, used for API access and references |
-
-These give you a fast read on scope without counting rows in the Traces tab.
-
-## Deep Analysis
-
-A single button that triggers on-demand investigation of the issue's representative trace.
-
-- **Idle**: a "Run Deep Analysis" button. Click to dispatch.
-- **Running**: a progress indicator with "Running analysis…". Takes about a minute. You can navigate away; analysis continues in the background.
-- **Complete**: "Analysis complete" with a Re-run option. Results populate the Overview tab's Probable Root Cause and Recommendations & Fixes.
-- **Failed**: "Retry Deep Analysis" if the last run failed.
-
-See [Deep Analysis](/docs/error-feed/features/deep-analysis) for when to use it and what to expect.
-
-## Timeline
-
-When the issue first appeared and when it was most recently seen.
-
-| Field | What it shows |
-|-------|---------------|
-| **First seen** | When Error Feed first detected this error pattern, relative to now |
-| **Last seen** | The most recent occurrence in the cluster |
-| **Age** | Days since first detection |
-
-A long age (e.g. 45 days) with a recent "last seen" means the issue has been around for a while without being resolved. Useful context for prioritization.
-
-## AI Metadata
-
-Trace-level context about the trace currently being viewed (the representative trace, unless you've picked a different one in the Traces tab).
-
-| Field | What it shows |
-|-------|---------------|
-| **Model** | The LLM model used in the trace |
-| **Version** | The model version |
-| **Agent** | The agent name, if set in trace attributes |
-| **Pipeline** | The pipeline name, if set |
-| **Connector** | The integration connector used |
-| **Project** | The Observe project this trace belongs to |
-| **Eval score** | The composite evaluation score for this trace |
-| **Trace ID** | The trace's unique identifier |
-
-Fields not set on the trace are omitted. AI Metadata is useful for correlating issues to specific model versions, e.g. confirming a degradation started when you switched model versions.
-
-
-AI Metadata reflects the selected trace. Click a different trace in the Traces tab to see its metadata here.
-
-
-## Evaluations
-
-Quality scores for the currently selected trace. Each evaluation is a named row with either:
-
-- A **score bar and percentage** for LLM-judge evaluations (e.g. Factual Grounding: 62%)
-- A **pass/fail verdict** with a green check or red X for pass/fail evaluations
-
-These correspond to the four dimensions in [Scoring](/docs/error-feed/concepts/scoring), plus any custom evaluations set up on the project.
-
-If every trace in a cluster shows Factual Grounding in the 20–40% range, something is systematically wrong with grounding even if the classifier didn't flag a specific hallucination.
-
-## Co-occurring Issues
-
-When multiple distinct clusters tend to appear together in the same traces, they show up here. Each entry has:
-
-- The co-occurring issue title
-- How many traces are shared between the two issues
-- Co-occurrence percentage (what fraction of this issue's traces also appear in that cluster)
-
-High co-occurrence (70%+) usually means a shared root cause: fixing one often fixes the other, or at minimum they're worth investigating together.
-
-Click any co-occurring issue to jump to its detail page.
-
-## Activity
-
-A timeline of significant events for this issue, starting with first detection. Status changes, assignment events, and comments will also land here as the issue moves through triage.
-
-## Integrations
-
-Connected issue-tracking tools. Currently **Linear** is supported.
-
-- Linear not connected: shows a "Connect" button that takes you to Settings → Integrations.
-- Connected with no ticket: shows a "Create issue" button.
-- Ticket already exists: shows the ticket ID with an option to open it.
-
-See [Linear Integration](/docs/error-feed/features/linear-integration) for setup and workflow details.
-
-## Next Steps
-
-
-
- The Overview tab the metadata panel sits next to.
-
-
- How to use the panel to move issues through triage.
-
-
diff --git a/src/pages/docs/error-feed/features/sampling.mdx b/src/pages/docs/error-feed/features/sampling.mdx
deleted file mode 100644
index 536e8abe..00000000
--- a/src/pages/docs/error-feed/features/sampling.mdx
+++ /dev/null
@@ -1,76 +0,0 @@
----
-title: "Error Feed Sampling: Controlling Trace Analysis Rate"
-description: "How sampling rate controls what percentage of traces Error Feed analyzes, and how to configure it per project in Observe settings."
----
-
-## About
-
-Error Feed doesn't analyze every trace by default. The **sampling rate** controls what percentage of incoming traces get analyzed. The tradeoff is coverage vs. cost: analyze more traces and you catch more errors, but you pay more for it.
-
-## Why sampling exists
-
-Production agents can produce a lot of traces. Analyzing 100% of them at all times gets expensive at scale. Sampling lets you dial in a rate that makes sense for your situation: full coverage during development, a reduced rate in production, or 100% for critical projects where nothing can be missed.
-
-The rate applies to new traces. Previously analyzed traces aren't affected when you change it.
-
-## How to configure sampling
-
-Sampling is configured per project in Observe settings.
-
-
-
- Navigate to your project in the Observe section of the Future AGI dashboard.
-
-
- Click the **Configure** (gear) icon in the project header to open the settings drawer.
-
-
- Find the **Error Feed sampling rate** control in the drawer. Drag the slider right to increase coverage, left to decrease.
-
-
- Click **Update** to apply. The new rate kicks in for traces that arrive after the update.
-
-
-
-
-The new rate only applies to new traces. Previously analyzed traces aren't re-analyzed or de-analyzed when you change the rate.
-
-
-## Choosing a sampling rate
-
-There's no universally correct rate. It depends on your trace volume, cost tolerance for the project, and how critical full error coverage is.
-
-| Situation | Recommended rate |
-|-----------|-----------------|
-| Development or testing | 100% — catch everything while you're actively iterating |
-| Low-volume production | 100% or close to it — the absolute cost is low |
-| High-volume production | 10–20% — enough to catch systematic issues, affordable at scale |
-| Critical path / safety-sensitive | 100% — can't afford to miss errors |
-| Cost-constrained, high volume | 5–10% — catches recurring patterns even at low rates |
-
-At 10% sampling, a systematic error that hits every trace shows up as a cluster with 10% of its true occurrence count. The error still gets detected and surfaced. Sampling reduces counts and may miss rare one-off failures, but it reliably catches recurring patterns.
-
-
-Start at 100% when you first set up a project. Once you understand the error landscape and have addressed the biggest issues, drop the rate to something that makes sense for your production volume.
-
-
-## Effect on cluster trace counts
-
-The trace count on an issue reflects how many analyzed traces ended up in the cluster, not the total number of traces where that error might have occurred. At 20% sampling, a cluster with 50 traces likely represents around 250 actual occurrences.
-
-Keep the sampling rate in mind when comparing cluster sizes across projects or time periods. A 100-trace cluster from a 10%-sampled project represents more actual errors than a 100-trace cluster from a 100%-sampled project.
-
-## Effect on new issue detection
-
-Rare errors (the ones that show up in only a small fraction of traces) are more likely to be missed at low rates. If you're hunting an edge case that only triggers occasionally, temporarily bump the sampling rate up for the duration of the investigation.
-
-## Next Steps
-
-
-
- How sampled traces become clusters and issues.
-
-
- View issues created from sampled traces.
-
-
diff --git a/src/pages/docs/error-feed/features/state-graph.mdx b/src/pages/docs/error-feed/features/state-graph.mdx
deleted file mode 100644
index 42827f43..00000000
--- a/src/pages/docs/error-feed/features/state-graph.mdx
+++ /dev/null
@@ -1,69 +0,0 @@
----
-title: "Error Feed State Graph: Agent Decision Flow Diagram"
-description: "How to read the State Graph tab, the agent decision flow diagram showing where traces diverge between success and failure paths."
----
-
-## About
-
-The **State Graph** tab visualizes how an agent moves through its workflow and where it fails. It surfaces structural patterns that would take a lot of reading to extract from raw trace data.
-
-
-
-## Agent Decision Flow
-
-A branching diagram mapping the paths an agent takes from invocation to completion:
-
-- **Shared steps**: the common entry path every trace follows (invocation, initial LLM call, etc.)
-- **Fork point**: where traces diverge into success and failure paths
-- **Failure branch**: the steps failing traces take, colored red
-- **Success branch**: the steps passing traces take, colored green
-- **Edge labels**: percentages on the fork edges showing what fraction of traces take each path
-
-Read left to right. Steps are nodes; transitions between steps are edges. A step that consistently appears only on the failure path is a strong root-cause candidate.
-
-
-
-### Reading the fork percentages
-
-The numbers on the fork edges are the most useful signal. If 92% of traces go down the failure path after a particular step, that step is where the problem concentrates. No need to reason about edge cases; the data is telling you where to look.
-
-A near-even split (50/50) means the problem depends on input characteristics, not a systematic code issue. A lopsided split (95/5) means a near-universal failure, probably a configuration or logic error.
-
-### Node types
-
-Different step types appear as visually distinct nodes:
-
-| Node type | What it represents |
-|-----------|-------------------|
-| Invocation | The agent entry point |
-| Agent run | An agent execution step |
-| LLM call | A language model inference step |
-| Tool execution | A tool or function call |
-| Evaluation | An inline quality check step |
-| Error | A step that resulted in an error state |
-| Success | A step that completed successfully |
-
-## When the State Graph is most useful
-
-The State Graph is most useful when:
-
-- The cluster has a moderate-to-large trace count (enough to make the percentages meaningful)
-- The agent has multiple steps (single-step agents don't produce interesting flow diagrams)
-- You suspect the failure is structural, tied to a specific workflow path, rather than random
-
-For small clusters or very simple agents, the [Overview tab](/docs/error-feed/features/issue-overview) is usually enough. The State Graph earns its keep when you're looking at a large cluster and want to understand the shape of failure before diving into individual traces.
-
-## Relationship to the Overview tab's Agent Flow
-
-The [Overview tab](/docs/error-feed/features/issue-overview) also has an Agent Flow section, but it's based on the representative trace and shows the narrative flow for that single trace. The State Graph is based on the whole cluster and shows the aggregate picture. Use Agent Flow to understand the specific failure; use State Graph to understand how widespread and how structural it is.
-
-## Next Steps
-
-
-
- The Overview tab where Agent Flow first appears.
-
-
- Jump from a State Graph node to the underlying trace.
-
-
diff --git a/src/pages/docs/error-feed/features/the-feed.mdx b/src/pages/docs/error-feed/features/the-feed.mdx
deleted file mode 100644
index d7e7ec04..00000000
--- a/src/pages/docs/error-feed/features/the-feed.mdx
+++ /dev/null
@@ -1,99 +0,0 @@
----
-title: "The Error Feed Issue List: Filters, Stats, and Columns"
-description: "How to read the Error Feed issue list: filters, the stats bar, table columns, trend sparklines, and time range controls."
----
-
-## About
-
-The Feed is the landing page for Error Feed, in the left sidebar under **Error Feed**. It shows every detected issue across your projects as a filterable, sortable list. This is where you start every triage session.
-
-
-
-## Filter bar
-
-The filter bar at the top narrows the list to what matters right now.
-
-| Filter | Options |
-|--------|---------|
-| **Project** | Select one or more Observe projects. Defaults to all projects. |
-| **Time range** | Last 24 hours / 7 days / 14 days / 30 days / 90 days |
-| **Status** | Unresolved, Acknowledged, Resolved, Escalating |
-| **Severity** | Critical, High, Medium, Low |
-
-Filters combine. Selecting **Critical + Unresolved** shows only critical issues that haven't been acted on.
-
-
-Start every session with **Unresolved + Critical** to see the highest-priority uninvestigated issues first.
-
-
-## Stats bar
-
-The stats bar below the filter gives a snapshot of the current view:
-
-- **Total issues**: distinct clusters matching the current filters
-- **Total occurrences**: individual trace errors across those clusters
-- **New issues**: clusters that first appeared within the selected time range
-
-These update as you change filters. Use the occurrence count to gauge scope: a cluster with 5 traces is very different from one with 500.
-
-## Issue table
-
-Each row in the table represents one issue (cluster). The columns are:
-
-### Error name
-
-A human-readable name for the cluster, generated from the error pattern. This is what you read to figure out what's going wrong — e.g. "Tool parameter validation failed on search_docs" or "Hallucinated product availability".
-
-### Severity
-
-Critical (red), High (orange), Medium (yellow), Low (gray). See [Severity and Status](/docs/error-feed/concepts/severity-and-status) for what each level means and how to change it.
-
-### Status
-
-Current triage status: Unresolved, Acknowledged, Resolved, or Escalating. Issues you haven't looked at yet stay Unresolved.
-
-### Traces
-
-Number of individual traces grouped into this cluster. A high trace count means the error is recurring frequently.
-
-### Trend
-
-A sparkline showing how often this error appeared over the selected time range. Upward trend, getting worse. Flat, stable. Downward, resolving on its own (or you fixed something upstream).
-
-A directional indicator next to the sparkline shows the same thing at a glance.
-
-## Clicking an issue
-
-Click any row to open the issue detail page. It has four tabs (Overview, Traces, State Graph, Trends) and a metadata panel on the right.
-
-See the feature pages for each:
-
-
-
- Description, root cause, evidence, and recommendations.
-
-
- All traces in the cluster.
-
-
- Visual breakdown of where and how the agent failed.
-
-
- Error frequency, score trends, and activity heatmap.
-
-
-
-## Navigation tip
-
-The list remembers your last filter state within a session. Open an issue, hit the breadcrumb to go back, and your filters are still applied.
-
-## Next Steps
-
-
-
- Description, root cause, evidence, and recommendations.
-
-
- Move issues from new to resolved efficiently.
-
-
diff --git a/src/pages/docs/error-feed/features/traces.mdx b/src/pages/docs/error-feed/features/traces.mdx
deleted file mode 100644
index 27d37103..00000000
--- a/src/pages/docs/error-feed/features/traces.mdx
+++ /dev/null
@@ -1,59 +0,0 @@
----
-title: "Error Feed Traces Tab: Navigating Failure Clusters"
-description: "How to use the Traces tab on an issue detail page to navigate every trace in a cluster and understand the distribution of failures."
----
-
-## About
-
-The **Traces** tab lists every individual trace grouped into the issue's cluster. The tab label shows the count, e.g. "Traces 47" means 47 traces have been grouped under this issue.
-
-
-
-## What you see
-
-Each row is one trace from the cluster:
-
-- **Trace ID**: unique identifier you can use to look it up in Observe
-- **Status**: whether this trace was a failure or (where applicable) a comparison success trace
-- **Timestamp**: when the trace happened
-- **Latency, tokens, and cost metadata**: where available on the original trace
-
-The tab header shows the total count so you can size up the cluster without scrolling.
-
-## Navigating between traces
-
-Clicking a trace selects it as the active trace for the detail view. Two effects:
-
-1. The **metadata panel** on the right updates to show AI Metadata and Evaluations for the selected trace instead of the representative trace.
-2. The **Deep Analysis** button in the metadata panel will run against the selected trace if triggered.
-
-Useful for investigating specific traces within a cluster: comparing a trace from three days ago against a recent one, or pulling up a trace from a particular user or session.
-
-
-The Overview tab always shows analysis for the cluster's representative trace, not whichever one you've selected in the Traces tab. Check the metadata panel's AI Metadata section to confirm which trace is being analyzed.
-
-
-## Using the Traces tab to understand scope
-
-The trace count on the tab label is the most direct signal of how widespread the problem is. One trace might be a one-off; 500 traces is hitting a large fraction of your traffic (relative to sampling rate).
-
-For high-count clusters, check whether the traces are clustered in time (a systemic issue that appeared on a specific day) or spread evenly over weeks (a recurring edge case, not a single event).
-
-The [Trends tab](/docs/error-feed/features/trends) gives you the time-series view, which is the better tool for that kind of temporal analysis.
-
-## Relationship to the Observe trace view
-
-The Traces tab is specific to Error Feed and only shows traces belonging to this cluster. It doesn't replace the Observe trace view: no complete span tree, no annotation editing.
-
-For span-level inspection of a specific trace, copy the Trace ID and look it up directly in Observe.
-
-## Next Steps
-
-
-
- Description and analysis for the selected trace.
-
-
- Visualize where in the workflow traces fail.
-
-
diff --git a/src/pages/docs/error-feed/features/trends.mdx b/src/pages/docs/error-feed/features/trends.mdx
deleted file mode 100644
index 772604c7..00000000
--- a/src/pages/docs/error-feed/features/trends.mdx
+++ /dev/null
@@ -1,72 +0,0 @@
----
-title: "Error Feed Trends: Score Trends and Activity Heatmap"
-description: "How to use the Trends tab (Events Over Time, Score Trends, and the Activity Heatmap) to understand how an issue is evolving."
----
-
-## About
-
-The **Trends** tab shows the temporal story. Is this getting worse? Did it spike after a deployment? What time of day does it concentrate? Are scores improving or degrading?
-
-
-
-## Events Over Time
-
-How often errors in this cluster have occurred over the selected time range. Two series:
-
-- **Errors** (area/line): error occurrences per day in this cluster
-- **Traffic** (bars): total trace volume Error Feed analyzed in the same period
-
-Showing traffic alongside errors is deliberate. A higher error count might just mean more traffic; the actual error *rate* could be stable. When the bars (traffic) and the line (errors) rise together proportionally, the rate is holding steady. When errors outpace traffic growth, the problem is genuinely getting worse.
-
-
-
-### Reading spikes
-
-A sudden spike on a specific day is worth correlating with your deployment history. If you shipped a new model, updated a prompt, or changed tool configurations that day, the spike likely traces back to that change.
-
-A gradual upward slope (rather than a spike) means the error is tied to changing input distribution: the kinds of queries your users send are shifting in a direction that triggers this failure mode more often.
-
-## Score Trends
-
-How the four quality dimension scores (Factual Grounding, Privacy & Safety, Instruction Adherence, Optimal Plan Execution) have moved over time for traces in this cluster.
-
-Each dimension is a line chart. Declining line, quality is degrading. Rising line, improving. Flat means consistent, which could be consistently good or consistently bad depending on the absolute level.
-
-Most useful for tracking whether a deployed fix actually improved quality. After resolving an issue and deploying a change, watch Score Trends over the next few days to confirm the affected dimension is moving up.
-
-## Activity Heatmap
-
-A grid of error frequency by hour of day and day of week. Each cell is a specific hour on a specific day; darker cells mean higher error counts.
-
-
-
-### What the heatmap tells you
-
-Patterns in the heatmap reveal whether the error is tied to usage. Common ones:
-
-- **Weekday mornings, low on weekends**: correlates with business-hours usage, likely triggered by specific user behavior rather than a random code bug
-- **Uniform distribution**: the error happens randomly across hours and days, so it's input-independent and truly systematic
-- **Specific hours**: a concentration at certain hours might correlate with a scheduled job, a batch process, or peak usage from a specific timezone
-
-These patterns help you tell whether the problem is urgent (consistent, all-hours) or a usage-pattern correlation that might resolve once the triggering input changes.
-
-## Combining the three views
-
-The most useful read is all three together. A typical investigation:
-
-1. **Events Over Time**: is the issue growing or shrinking? If growing, look at the rate relative to traffic.
-2. **Score Trends**: are any quality dimensions trending down? If Factual Grounding has been declining for a week, something upstream changed.
-3. **Heatmap**: is there a temporal pattern? Errors concentrating at 9am UTC every weekday is a clue about what triggers them.
-
-This combination often turns a confusing cluster into a clear, attributable pattern.
-
-## Next Steps
-
-
-
- Use trends to decide whether a fix held.
-
-
- How the four quality scores are computed.
-
-
diff --git a/src/pages/docs/error-feed/features/triage-workflow.mdx b/src/pages/docs/error-feed/features/triage-workflow.mdx
deleted file mode 100644
index ee9f3e5a..00000000
--- a/src/pages/docs/error-feed/features/triage-workflow.mdx
+++ /dev/null
@@ -1,106 +0,0 @@
----
-title: "Error Feed Triage Workflow: Resolve, Ignore, and Escalate"
-description: "How to move issues through the Error Feed triage workflow: resolving, acknowledging, ignoring, assigning, and escalating."
----
-
-## About
-
-Triage is reviewing new issues, deciding what to do with each one, and tracking them through to resolution. Error Feed is built to fit into your existing workflow rather than replace it: issues have statuses, assignees, and integrations that connect to whatever process you already use.
-
-## Status lifecycle
-
-An issue moves through four primary states:
-
-```
-Unresolved → Acknowledged → Resolved
- ↕
- Escalating
-```
-
-**Unresolved** is the starting state. Every new cluster starts here. Filter the feed to Unresolved to see what hasn't been looked at.
-
-**Acknowledged** means someone has reviewed the issue and confirmed it's real and worth tracking. It doesn't mean a fix is in progress; it means the issue has left the "inbox." Use Acknowledged to reduce noise so your team knows what's new vs. what's known.
-
-**Escalating** means the issue is actively getting worse or a previous fix didn't hold. Use this to flag urgency beyond "this is open." Issues can go straight from Unresolved to Escalating if they warrant immediate attention.
-
-**Resolved** means the fix is deployed and the issue is closed. Error Feed keeps monitoring; if the same cluster pattern reappears, it creates a new issue rather than reopening the old one. This keeps resolved issues genuinely resolved and regressions visible.
-
-## Changing status
-
-Three paths:
-
-**Header buttons**: the fastest route. The issue detail header has Resolve, Acknowledge, and Ignore issue buttons. One click, no confirmation.
-
-**Status dropdown in the metadata panel**: click the current status chip in the Triage section for a dropdown with all states. Use this for states like Escalating that don't have a dedicated header button.
-
-**Feed list**: status changes made in the detail view show up in the list view immediately. No reload needed.
-
-
-"Ignore issue" currently sets the status to Escalating, which doesn't permanently suppress the issue. To stop seeing an issue, set it to Resolved.
-
-
-## Assigning issues
-
-Issues can be assigned to anyone in your organization. The assignee field is in the Triage section of the [metadata panel](/docs/error-feed/features/metadata-panel).
-
-Click **Assign** to open the picker, then pick a team member. To reassign, click the current assignee and pick a new one. To unassign, click the current assignee and select Unassign.
-
-Assignment is informational right now: it doesn't send a notification. For that, use the [Linear integration](/docs/error-feed/features/linear-integration) to create a ticket and assign it there.
-
-## A practical triage session
-
-Here's how to work through a backlog of issues efficiently:
-
-
-
- Start with the highest-severity issues nobody has looked at yet. Your "on fire right now" queue.
-
- In the Feed list, set Status to **Unresolved** and Severity to **Critical**.
-
-
- The [Overview tab](/docs/error-feed/features/issue-overview)'s Description section says what's going wrong in plain language. For most issues you'll know within 30 seconds whether it's a real problem or a false positive.
-
-
- Recognize the issue and already have a fix in flight: set it to **Acknowledged** and assign it to whoever owns the fix.
-
- Already fixed from a previous deploy: set it to **Resolved**.
-
- Not something you're going to act on: still Acknowledge it so others know it's been reviewed.
-
-
- Use the [Trends tab](/docs/error-feed/features/trends) for severity and trajectory. Use the [State Graph](/docs/error-feed/features/state-graph) to see where in the workflow the failure concentrates. Use [Deep Analysis](/docs/error-feed/features/deep-analysis) when an issue warrants more investigation.
-
-
- For issues that need a proper fix tracked in your project management tool, use the [Linear integration](/docs/error-feed/features/linear-integration) to create a ticket directly from the metadata panel. The ticket is linked to the cluster, so you can navigate between them.
-
-
- Work down the severity ladder. High after Critical, Medium after High. Low-severity issues can be batched into a weekly review instead of triaged one by one.
-
-
-
-## Handling escalating issues
-
-An issue escalates when the problem is getting worse. In practice: watch the [Trends tab](/docs/error-feed/features/trends) Events Over Time chart. If the error count is rising week over week, move the issue to Escalating.
-
-Treat Escalating issues like Critical regardless of their assigned severity. Severity is about the *type* of problem; Escalating is about the *trajectory*.
-
-## After a fix is deployed
-
-When you ship a fix for a resolved issue:
-
-1. Set the issue to **Resolved** if it isn't already.
-2. Come back 24–48 hours later and check the [Trends tab](/docs/error-feed/features/trends). Confirm Score Trends are improving and the Events Over Time count is dropping.
-3. If a new issue appears in the same category with a similar name, the fix may have partially worked, or a regression was introduced. Investigate the new cluster.
-
-The goal isn't an empty issue list. It's making sure nothing critical sits unreviewed and that resolved issues actually stay resolved.
-
-## Next Steps
-
-
-
- What each status and severity tier means.
-
-
- Create tickets from issues during triage.
-
-
diff --git a/src/pages/docs/error-feed/guides/create-linear-issue.mdx b/src/pages/docs/error-feed/guides/create-linear-issue.mdx
new file mode 100644
index 00000000..7f5c6890
--- /dev/null
+++ b/src/pages/docs/error-feed/guides/create-linear-issue.mdx
@@ -0,0 +1,50 @@
+---
+title: "Create a Linear issue"
+description: "The team picker creates the ticket the moment you click, and nothing syncs back afterwards"
+---
+
+Turning an Error Feed issue into a Linear ticket gets the fix into your engineering team's actual backlog instead of leaving it to sit in the Feed. This covers that one job: linking one issue to one new Linear ticket. Linear must already be connected for the workspace, under **Settings > Integrations**, before any of this works.
+
+## Create the ticket
+
+Open the issue from the [Feed](/docs/error-feed/guides/triage-issues) to land on its detail page, then scroll the [metadata sidebar](/docs/error-feed/guides/investigate-an-issue) to the Integrations section. The Linear row reads **Create issue** once the workspace is connected.
+
+Click it, and a **Create Linear Issue** dialog opens with the line "Select a team to create the issue in." followed by your Linear teams.
+
+
+Clicking a team creates the ticket immediately, with no confirm button. The link between the issue and that ticket is permanent and can't be removed from Error Feed, so check you've picked the right team before you click.
+
+
+Click the team you want the ticket filed under. The Linear row then reads **View** followed by the issue ID, for example ENG-1234, and clicking it opens the ticket in Linear. The ticket takes its title from the Error Feed issue, truncated at 200 characters.
+
+A toast confirms it went through: "Created" plus the issue ID. If it fails instead, the toast reads "Failed to create Linear issue" and the row stays on **Create issue** so you can retry.
+
+### One ticket per issue
+
+Only one Linear issue can be linked per Error Feed issue. Once one exists, the row shows **View** instead of **Create issue**, and there's no separate button to link a second ticket.
+
+## What the team picker can show instead
+
+The **Create Linear Issue** dialog can land on one of four states instead of a team list:
+
+| Message | What to do |
+|---|---|
+| "Loading teams" | Wait a moment for the fetch to finish |
+| "Couldn't reach Linear. Check the integration in Settings and try again." | Check the integration under Settings > Integrations and retry |
+| "Linear isn't connected for this workspace. Connect it in Settings > Integrations." | Connect it under Settings > Integrations |
+| "No teams found in your Linear workspace." | Add a team in your Linear workspace: there's none to file the ticket into |
+
+## Nothing syncs back
+
+Creating the ticket is one-directional. Closing, resolving, or otherwise updating the Linear ticket doesn't touch the Error Feed issue. The issue keeps whatever status it had when you created the ticket, so once the fix lands, resolve the Error Feed issue yourself.
+
+## Dive deeper
+
+
+
+ Where creating a Linear ticket fits into resolving, acknowledging, and assigning issues
+
+
+ Get a written root cause and a proposed fix for the cluster behind the issue
+
+
diff --git a/src/pages/docs/error-feed/guides/investigate-an-issue.mdx b/src/pages/docs/error-feed/guides/investigate-an-issue.mdx
new file mode 100644
index 00000000..5c7a1ac5
--- /dev/null
+++ b/src/pages/docs/error-feed/guides/investigate-an-issue.mdx
@@ -0,0 +1,90 @@
+---
+title: "Investigate an issue"
+description: "Read an issue's header and sidebar, then route between Overview, Traces, and Trends to confirm what's going wrong."
+---
+
+Open an issue from the [Feed](/docs/error-feed/guides/triage-issues) list and you land on its detail page: a header, a metadata sidebar, and a tab bar with **Overview**, **Traces**, **Trends**, and **Fix**, all views onto the same [cluster](/docs/error-feed/concepts/understanding-error-feed) of traces that failed the same way. Fix has its own guide; this one covers the other three.
+
+This guide works the detail page in the order that actually finds a problem: orient in the header, read the pattern on Overview, evidence the divergence with the trace evidence reel and split compare, drop into Traces only if the pattern doesn't hold up, and check Trends for how urgent it is.
+
+read the pattern-summary cards"] --> B{"One consistent failure mode?"}
+ B -->|"yes"| C["Evidence it evidence reel + split compare"]
+ B -->|"no, or need one run"| D["Traces tab find the specific run"]
+ C --> E["Trends tab how urgent is it?"]
+ D --> E`} />
+
+## Header and sidebar
+
+The header and the right-hand sidebar stay fixed across every tab. The header carries:
+
+- A breadcrumb and error-type chip
+- The issue title
+- Status and severity badges
+- A trace-count chip
+
+The sidebar holds status, severity, and assignee as editable controls. See [Triage issues](/docs/error-feed/guides/triage-issues) for how to use them.
+
+Use **Copy cluster ID** to paste the cluster identifier into a ticket or message, and **Share** for a direct link to this issue. See [Issue fields & filters](/docs/error-feed/reference/issue-fields) for what everything on the header and sidebar means.
+
+## Overview tab: is this one failure or several?
+
+Overview opens by default, and it's where every investigation starts: work out whether the cluster is one clean failure or several tangled together before you dig into individual traces.
+
+
+*The Overview tab: pattern-summary cards, the events-and-users chart, and the trace evidence reel*
+
+### Read the pattern
+
+The pattern-summary cards describe what's common across the whole cluster, not just one trace. Read them first: if they point at one consistent failure mode, you're likely looking at a single clean cluster; if they point in different directions, the cluster may be mixing more than one failure mode and needs a closer, trace-by-trace look.
+
+A chart below the cards plots events and users for the cluster, so you can see whether it's a steady trickle or a recent spike.
+
+### Open the evidence reel
+
+The trace evidence reel is on the Overview tab, with a switcher above it for its three view modes:
+
+- **Breadcrumb**: a linear read of what happened, the one to reach for first
+- **Agent Graph**: every step the agent could take, useful for seeing whether the failure sits on one path among several or is the agent's only option
+- **Agent Path**: the sequence this particular run actually took, useful for tracing exactly where this one run went sideways
+
+Within the reel, two tabs separate the evidence: **Failing** shows one failing trace at a time from those backing the pattern, and **Working** shows the nearest trace that succeeded.
+
+### Split-compare to find the divergence
+
+Toggle **Split with working** to line the open failing trace up against that nearest working trace (toggle **Single view** to go back to one trace at a time). This pairing is matched ahead of time by Error Feed, not a random working trace picked on the spot, so it's built to show exactly where the two runs diverge.
+
+Not every cluster has a working trace to pair against. If none was found, split compare has nothing to show; work from the Traces tab instead.
+
+## Traces tab: find the specific run
+
+If Overview's pattern doesn't hold up under a closer look, or you need one specific run rather than the aggregate, drop into the Traces tab: it lists the cluster's traces, one row each.
+
+Five aggregate cards sit at the top: **Total traces**, **Avg score**, **Avg turns**, **P50 latency**, **P95 latency**. The grid below carries a column for each: **Trace ID**, **Input**, **Start Time**, **Duration**, **Tokens**, **Cost**, **Score**. Click any row to open it in the trace drawer for the full detail.
+
+
+Voice and simulator projects open a different trace drawer here. See [Voice observability](/docs/observe/features/voice) and [Explore results](/docs/simulation/guides/explore-results).
+
+
+## Trends tab: is this urgent?
+
+Trends is a single chart: errors and traffic plotted on two axes over time. Read the two lines as a pair, not separately. If the error line climbs while traffic barely moves, something got worse in the system itself. If both climb together, you're most likely looking at more volume, not a rising failure rate, which is often enough on its own to tell you whether an issue is urgent or just a side effect of growth.
+
+## Dive deeper
+
+
+
+ Change status, severity, and assignee once you know what's wrong
+
+
+ Get a written root cause and a proposed fix from the Fix tab
+
+
+ Turn the finding into a ticket your team can work from
+
+
+ The full list of columns, cards, and values referenced on this page
+
+
diff --git a/src/pages/docs/error-feed/guides/run-root-cause-analysis.mdx b/src/pages/docs/error-feed/guides/run-root-cause-analysis.mdx
new file mode 100644
index 00000000..4591eea1
--- /dev/null
+++ b/src/pages/docs/error-feed/guides/run-root-cause-analysis.mdx
@@ -0,0 +1,34 @@
+---
+title: "Run a root cause analysis"
+description: "Get a written root cause and a proposed fix for one cluster from the Fix tab."
+---
+
+The **Fix** tab is Error Feed's agentic root-cause chat thread, with a follow-up composer underneath it. Start it and sub-agents sample representative calls from the [cluster](/docs/error-feed/concepts/understanding-error-feed), compare them against a passing baseline, and synthesise a written root cause and a proposed fix in the thread. This page walks through starting a run, reading what comes back, asking a follow-up, and re-running it later.
+
+## Run the analysis
+
+Start from an issue's detail page, with the failure pattern already confirmed via [Investigate an issue](/docs/error-feed/guides/investigate-an-issue).
+
+From the issue's detail page, click into the Fix tab. If nothing has run yet it shows an empty state, **No analysis yet**, with a button labeled **Analyze this cluster**. You can also start it from the issue's headline card, whose button reads **Debug this cluster** before a run exists. Both start the same run, and each uses 1 credit, taken when the run starts and refunded if the run fails.
+
+
+*The Fix tab's empty state, before any run has started*
+
+The run starts immediately and the tab keeps checking until it lands or fails, with a one-hour cut-off. The result arrives as a single written message in the thread: a root cause explaining what's going wrong, and a proposed fix for it, based on the calls Falcon sampled. To probe the reasoning further, type into the composer at the bottom (placeholder: **Ask Falcon a follow-up...**), and **Falcon is investigating...** shows while a reply streams in.
+
+Once the cluster has picked up new traces, click **Re-run** in the tab's header (tooltip: **Re-run with current cluster state (1 credit)**) to analyze it against its current state. The headline card carries the same option once a run exists, tooltipped **Re-run analysis (1 credit)**.
+
+## If it fails
+
+A run can fail with one of two messages: **Couldn't start the analysis. Please try again.** or **Couldn't connect to the server. Please try again.** For what each one means and what to do about it, see [Analysis doesn't finish](/docs/error-feed/troubleshooting/analysis-does-not-finish).
+
+## Dive deeper
+
+
+
+ Turn the finding into a ticket your team can work from
+
+
+ What to check when a run stalls or never lands
+
+
diff --git a/src/pages/docs/error-feed/guides/triage-issues.mdx b/src/pages/docs/error-feed/guides/triage-issues.mdx
new file mode 100644
index 00000000..f8320629
--- /dev/null
+++ b/src/pages/docs/error-feed/guides/triage-issues.mdx
@@ -0,0 +1,75 @@
+---
+title: "Triage issues"
+description: "Narrow a full feed to what's worth acting on, then resolve, acknowledge, or reassign the issues that matter."
+---
+
+Error Feed's [list page](/docs/error-feed/guides/triage-issues), in the left sidebar under **Error Feed** (see the [overview](/docs/error-feed) if you haven't opened it yet), shows every detected issue across your projects, scoped to the last 7 days until you change the range. A full feed is a queue, not a to-do list: some rows need attention today, most don't.
+
+This guide takes you from a full feed to a handled list: narrow it to what's worth looking at, scan the table for what actually decides priority, then act on what you find, one issue at a time or several at once.
+
+## Narrow the feed
+
+Type into the search box (placeholder **Search errors**) to match against the error name, issue group, or category. Next to it sit five selects: project, status, severity, and fix layer each open on an All value until you narrow them, while time range opens already scoped to Last 7 days. See [Issue fields & filters](/docs/error-feed/reference/issue-fields) for the exact set of values each one accepts.
+
+
+Start from **Severity: Critical** and **Status: Escalating**. Critical narrows to what matters most, and Escalating is the one status of the four that hasn't settled into Acknowledged, For review, or Resolved. See [Severity & Status](/docs/error-feed/concepts/severity-and-status) for when to pick each one.
+
+
+Once project, status, severity, or fix layer is set, a **Clear** button appears next to the selects (tooltip: "Clear all filters") to reset everything in one click instead of undoing each select by hand.
+
+## Scan the table
+
+Each row is one issue. Eight columns run left to right: **Error**, **Severity**, **Status**, **Events**, **Users**, **Fix Layer**, **Trend (14d)**, and **Last seen**.
+
+Three of them decide priority. **Severity** says how bad it is, **Status** says where it sits in your workflow, and **Trend (14d)** shows whether it's climbing, flat, or settling down.
+
+The rest is context:
+
+- **Error** names what's failing
+- **Events** and **Users** size the blast radius
+- **Fix Layer** points at where in your system the fix belongs
+- **Last seen** says when it last fired
+
+At the bottom, set **Results per page** to 10, 25, or 50, and move through the rest with **Back** and **Next**.
+
+## Act on what you find
+
+Three ways to act, each suited to a different job in the [triage workflow](/docs/error-feed/guides/triage-issues):
+
+- **Bulk actions** for many rows moving to the same status at once
+- **Header buttons** for a quick resolve or acknowledge on a single issue
+- **Metadata sidebar** for a severity or assignee change on a single issue
+
+
+None of the three ways below show a toast. There's no confirmation of success and no warning on failure, so the save happens silently either way. To check a change went through, re-check the **Status** (or **Severity**) column for that row, or refresh the feed; if the value hasn't moved, repeat the action.
+
+
+### Handle many at once
+
+Tick the checkbox on any row and a bulk-select toolbar appears above the table. Tick more rows, then open **Bulk actions** and pick **Mark as Resolved**, **Mark as Acknowledged**, **Mark as For Review**, or **Mark as Escalating** to move every selected issue to that status in one go.
+
+### Resolve or acknowledge one issue
+
+Click a row to open the issue. Its header carries three buttons: **Resolve** moves the issue to resolved, **Acknowledge** moves it to acknowledged, and **Ignore issue** moves it to escalating despite the label. All three disable themselves while the update is in flight.
+
+### Change status, severity, or assignee from the sidebar
+
+Every issue also has a metadata sidebar on the right. Its **Status** and **Severity** rows each open a menu to set a new value directly. The **Assignee** row reads **Assign** until someone's on it; click it to open a menu headed **Assign to**, listing everyone in your org plus an **Unassign** option once someone's set.
+
+
+Assigning someone to an issue only records it on the issue itself. It sends no notification of any kind, so tell them yourself if it needs to reach them.
+
+
+## Dive deeper
+
+
+
+ Read the evidence behind one issue before you touch the fix
+
+
+ The two independent axes every issue carries, and how they change
+
+
+ Every filter, column, and enum value in the feed
+
+
diff --git a/src/pages/docs/error-feed/guides/turn-on-error-feed.mdx b/src/pages/docs/error-feed/guides/turn-on-error-feed.mdx
new file mode 100644
index 00000000..d421201a
--- /dev/null
+++ b/src/pages/docs/error-feed/guides/turn-on-error-feed.mdx
@@ -0,0 +1,53 @@
+---
+title: "Turn on Error Feed"
+description: "Turn on trace scanning for a project so Error Feed can find its first issues."
+---
+
+Error Feed ships off. A project's scanner starts at a 0% sampling rate, so nothing gets scanned until you raise it. This guide gets a project from silent to its first issues showing up in the Feed.
+
+## Before you start
+
+- The project is already receiving traces in [Observe](/docs/observe). Error Feed only scans what Observe receives, so [send your agent through a request](/docs/observe/quickstart) first if none have arrived yet
+- Your workspace needs the Error Feed capability; without it, the Error Feed page shows the upgrade message and the API answers 402
+
+## Turn on scanning
+
+Open the Observe project you want issues for, then click the settings gear icon, tooltipped **Settings**, in the project header. A drawer titled **Configure Project** opens, carrying the project's settings including sampling.
+
+Find the sampling rate control in that drawer and raise it above 0. 100% is a safe default while you're trying it out; see [Choosing a rate](#choosing-a-rate) for the cost tradeoff once volume climbs. Click **Update** to apply it.
+
+Then wait before checking for results, because a new rate only reaches traces that arrive after you save it. Send your agent through a request, give scanning a moment, and go to **Error Feed** in the left sidebar. A row appearing in the list confirms scanning is live. An upgrade prompt instead of the Feed means the workspace lacks the Error Feed capability.
+
+## Choosing a rate
+
+There's no universally right number; it's a coverage-versus-cost call: analyze more traces and you catch more, but you pay more for it.
+
+| Situation | Rate |
+|-----------|------|
+| Building or testing a project | 100%, so nothing slips past you |
+| Low-volume production | 100%, the absolute cost stays low |
+| High-volume production | 10–20%, enough to catch recurring issues |
+| Cost-constrained, high volume | 5–10%, still catches patterns that repeat |
+
+A rate change only reaches forward. It applies to traces that arrive after you save it, not to anything that already went by.
+
+## When scanning runs
+
+Scanning is triggered per trace, not on a timer. Once a trace's root span completes, Error Feed waits about ten seconds before a scan starts and samples it.
+
+Traces that arrive through the [collector](/docs/error-feed/troubleshooting/no-issues-in-the-feed#the-traces-came-in-through-the-collector) instead of the inline path skip that trigger. A periodic sweep picks them up instead, working through them in small batches rather than the moment they land. If your traces go through the collector, expect the first issues to show up in occasional bursts rather than a steady trickle.
+
+## If nothing shows up
+
+If you've confirmed traces are reaching the project and the rate is saved, see [No issues in the Feed](/docs/error-feed/troubleshooting/no-issues-in-the-feed) for the full list of causes, in order.
+
+## Dive deeper
+
+
+
+ The mental model: how a sampled trace becomes an issue in the Feed
+
+
+ Where your first issues show up, and how to read the list
+
+
diff --git a/src/pages/docs/error-feed/index.mdx b/src/pages/docs/error-feed/index.mdx
index 23a85838..ecf1c995 100644
--- a/src/pages/docs/error-feed/index.mdx
+++ b/src/pages/docs/error-feed/index.mdx
@@ -1,74 +1,39 @@
---
-title: "Future AGI Error Feed: AI Agent Trace Error Detection"
-description: "Automatically detect, cluster, score, and triage errors in your AI agent traces, without any configuration beyond standard tracing."
+title: "Overview"
+description: "Error Feed reads your traces, groups the problems it finds into issues, and points at the layer to fix"
---
-## About
+Errors in an AI system rarely show up as one clean failure. They show up as a pattern: the same kind of mistake repeating across dozens of requests, buried in traces you'd otherwise have to read one by one to catch.
-Error Feed is Future AGI's error monitoring for AI agents. As soon as traces hit an Observe project, it picks them up, finds failure patterns, groups similar ones together, and writes up the analysis. No extra setup.
+## What is Error Feed?
-Think Sentry, but for the ways agents actually fail: hallucinated outputs, tool misuse, broken workflows, safety violations, and reasoning gaps that traditional error monitoring won't catch.
+**Error Feed** reads a sample of the traces in an [Observe](/docs/observe) project, decides for itself what went wrong in each one, and groups the traces that went wrong the same way into a single issue you work like a ticket. Nothing has to mark a trace as failed first: the scan is what finds the problem. Each issue carries a severity, a status, an assignee, and an optional link to a [Linear](/docs/error-feed/guides/create-linear-issue) ticket, and points at the fix layer, the part of your system the fix actually belongs in.
-
+## Before you start
-## What it does
+Error Feed requires an Enterprise or Cloud license.
-Error Feed runs in the background on every Observe project. For each trace it analyzes, it:
+Scanning ships off because a project's sampling rate starts at 0. Nothing is scanned, and no issues appear, until you [raise it](/docs/error-feed/guides/turn-on-error-feed).
-- **Detects errors** in five categories, from factual grounding failures to tool crashes to safety violations. See the full [error taxonomy](/docs/error-feed/concepts/taxonomy).
-- **Groups related traces** into named clusters, so 50 traces with the same underlying problem show up as one issue instead of 50 alerts.
-- **Scores the trace** on four quality dimensions, each on a 0–5 scale. See [Scoring](/docs/error-feed/concepts/scoring).
-- **Generates analysis**: what went wrong, root causes, supporting evidence from the trace, plus a quick fix and a long-term recommendation.
-- **Tracks trends**: whether an issue is happening more often, less often, or staying steady.
+## Start here
-
-No configuration needed. Error Feed turns on automatically for any Observe project the moment traces start arriving.
-
-
-## Who it's for
-
-Useful whether you're debugging an agent that just started misbehaving, doing a quality review, or trying to spot systemic problems across thousands of production traces.
-
-You don't need to know how transformers work to use it. The UI explains what went wrong in plain language. If you do want to dig into trace-level evidence, every finding links straight to the spans involved.
-
-## Supported integrations
-
-Error Feed works with any integration that sends traces to a Future AGI Observe project.
-
-**LLM providers**: OpenAI, OpenAI Agents SDK, Vertex AI (Gemini), AWS Bedrock, Mistral AI, Anthropic, Groq, Together AI, Google ADK, Google GenAI, Portkey
-
-**Orchestration frameworks**: LlamaIndex, LlamaIndex Workflows, LangChain, LangGraph, LiteLLM, CrewAI, Haystack, Autogen, PromptFlow, Vercel, Pipecat
-
-**Other**: DSPy, Guardrails AI, Hugging Face smolagents, Ollama, Instructor, MCP
-
-## Navigate the docs
-
-
-
- The mental model: how traces become issues, clusters, and scored findings.
-
-
- The five error categories and every subcategory Error Feed can detect.
+
+
+ Raise the sampling rate on a project and get your first issues
-
- The four quality metrics, what they measure, and how to read scores.
+
+ The object model behind an issue, from a single finding to the fix layer
-
- Severity tiers and the triage status workflow — from new issue to resolved.
-
-
-
-
-
- Filters, stats bar, columns, sparklines — the issue list page.
+
+ The two independent axes every issue carries, and how they change
-
- The Overview tab: description, root cause, evidence, and recommendations.
+
+ Filter, scan, and work a feed down with bulk actions
-
- On-demand deeper analysis for issues that need more investigation.
+
+ Read the evidence behind one issue before you touch the fix
-
- Resolve, acknowledge, assign — how to move issues through your process.
+
+ Get a written root cause and a proposed fix for one issue
diff --git a/src/pages/docs/error-feed/reference/error-taxonomy.mdx b/src/pages/docs/error-feed/reference/error-taxonomy.mdx
new file mode 100644
index 00000000..59721b44
--- /dev/null
+++ b/src/pages/docs/error-feed/reference/error-taxonomy.mdx
@@ -0,0 +1,68 @@
+---
+title: "Error taxonomy"
+description: "The fixed groups, categories, and fix layers every Error Feed finding is classified into"
+---
+
+## The shape of a finding
+
+Every scanner [finding](/docs/error-feed/concepts/understanding-error-feed) Error Feed writes carries three tags: a **group**, a **category** inside that group, and a **fix layer** (the part of your system the finding points at). Eval-sourced findings are tagged differently: the eval name stands in for group, and category is left unset. Fix layer is the only one of the three that's a live filter on the [feed](/docs/error-feed/guides/triage-issues), with options for All Fix Layers, Prompt, Tools, Orchestration, and Guardrails. See [Issue fields & filters](/docs/error-feed/reference/issue-fields) for the full list of fields and filters.
+
+The set is fixed, not free-form. A scanner finding always lands in exactly one row of the table below, and points at one of four fix layers.
+
+5 categories"] --> T["Tools"]
+ B["Context & Retrieval 4 categories"] --> P["Prompt"]
+ D["Output Quality 3 categories"] --> P
+ C["Planning & Goals 3 categories"] --> O["Orchestration"]
+ E["Infrastructure 5 categories"] --> G["Guardrails"]`} />
+
+## Fix layers
+
+- **Tools**: fix layer for Tool Failures findings
+- **Prompt**: fix layer for Context & Retrieval and Output Quality findings
+- **Orchestration**: fix layer for Planning & Goals findings
+- **Guardrails**: fix layer for Infrastructure findings
+
+## Groups, categories & fix layers
+
+Five groups organize twenty categories.
+
+| Group | Category | What it means | Fix layer |
+|---|---|---|---|
+| Tool Failures | Tool-related | The agent mishandled a tool or its result: asserted success after an error, passed wrong arguments, or guessed instead of calling | Tools |
+| Tool Failures | Tool Selection Errors | The agent picked the wrong tool, or no tool, for the task | Tools |
+| Tool Failures | Tool Output Misinterpretation | The agent misread or misused the result a tool returned | Tools |
+| Tool Failures | Formatting Errors | The agent's tool call or output didn't match the expected format | Tools |
+| Tool Failures | Language-only | A hallucination purely in language, no tool involved | Tools |
+| Context & Retrieval | Context Handling Failures | The agent lost, dropped, or mishandled context it was given | Prompt |
+| Context & Retrieval | Poor Information Retrieval | The agent retrieved information that was irrelevant, incomplete, or wrong | Prompt |
+| Context & Retrieval | Incorrect Memory Usage | The agent used stored memory incorrectly, including outdated or unrelated memory | Prompt |
+| Context & Retrieval | Unsupported Claim | The agent stated something that tool output or the end user's own input doesn't support | Prompt |
+| Planning & Goals | Task Orchestration | The agent sequenced or delegated steps incorrectly | Orchestration |
+| Planning & Goals | Goal Deviation | The agent drifted from the goal it was given | Orchestration |
+| Planning & Goals | Resource Abuse | The agent used excessive steps, calls, or resources to complete the task | Orchestration |
+| Output Quality | Instruction Non-compliance | The agent's output didn't follow the instructions it was given | Prompt |
+| Output Quality | Incorrect Problem Identification | The agent misunderstood or misidentified the problem it was asked to solve | Prompt |
+| Output Quality | Incomplete Response | The agent's response was absent, empty, or truncated | Prompt |
+| Infrastructure | Environment Setup Errors | The agent's runtime environment wasn't configured correctly | Guardrails |
+| Infrastructure | Resource Not Found | The agent tried to reach a resource that doesn't exist | Guardrails |
+| Infrastructure | Authentication Errors | The agent failed to authenticate with a required service | Guardrails |
+| Infrastructure | Timeout Issues | A call the agent depended on didn't complete in time | Guardrails |
+| Infrastructure | Service Errors | A service the agent depended on returned an error | Guardrails |
+
+
+Five groups map onto only four fix layers: Context & Retrieval and Output Quality both resolve to Prompt, so a poor retrieval and an incomplete response can carry the same fix layer even though they belong to different groups.
+
+
+## Keep exploring
+
+
+
+ Filter, sort, and triage issues in the list view
+
+
+ Every field and filter across the feed's UI and APIs
+
+
diff --git a/src/pages/docs/error-feed/reference/issue-fields.mdx b/src/pages/docs/error-feed/reference/issue-fields.mdx
new file mode 100644
index 00000000..b3eb8358
--- /dev/null
+++ b/src/pages/docs/error-feed/reference/issue-fields.mdx
@@ -0,0 +1,138 @@
+---
+title: "Issue fields & filters"
+description: "Filter options, table columns, field values, and limits for the Error Feed's issue list"
+---
+
+## Feed filters
+
+UI controls on the feed's filter bar that pick from a fixed set of values, with the exact options each one offers. The filter bar also carries a project select, scoped to your org's projects, and a free-text **Search errors** box; this table covers only the selects with a fixed set of options.
+
+| Filter | Options |
+|---|---|
+| Time range | Last 24 hours, Last 7 days, Last 14 days, Last 30 days, Last 90 days |
+| Status | All Statuses, see Status values below |
+| Severity | All Severities, see Severity values & priority below |
+| Fix layer | All Fix Layers, Prompt, Tools, Orchestration, Guardrails |
+
+## Sort keys & directions
+
+The `sort_by` values the feed list enforces, and the feed table column header that triggers each one. The feed table only sorts on Severity, Events, and Last seen, so `first_seen` and `error_count` aren't reachable from the UI at all.
+
+| `sort_by` | Column header | Default |
+|---|---|---|
+| `last_seen` | Last seen | Default |
+| `first_seen` | Not exposed in the UI | |
+| `error_count` | Not exposed in the UI | |
+| `unique_traces` | Events | |
+| `severity` | Severity | |
+
+`sort_dir` takes `asc` or `desc`, and defaults to `desc`.
+
+## Status values
+
+The full set of values an issue's `status` can hold, used by both the UI filter and the feed's `status` value. See [Severity & Status](/docs/error-feed/concepts/severity-and-status) for what each one means.
+
+| Status | Label |
+|---|---|
+| `escalating` | Escalating |
+| `for_review` | For review |
+| `acknowledged` | Acknowledged |
+| `resolved` | Resolved |
+
+## Severity values & priority
+
+The full set of values an issue's `severity` can hold, used by both the UI filter and the feed's `severity` value, and the `priority` value each is stored as.
+
+| Severity | Label | Stored as |
+|---|---|---|
+| `critical` | Critical | `urgent` |
+| `high` | High | `high` |
+| `medium` | Medium | `medium` |
+| `low` | Low | `low` |
+
+## Fix layers
+
+The fix layers a finding can point at; used by both the UI filter and the feed table's Fix Layer column. See [Error taxonomy](/docs/error-feed/reference/error-taxonomy) for the full group-to-category breakdown behind each one.
+
+| Fix layer |
+|---|
+| Prompt |
+| Tools |
+| Orchestration |
+| Guardrails |
+
+## Source values
+
+Where an issue's underlying [finding](/docs/error-feed/concepts/understanding-error-feed) came from, and what each source value means.
+
+| Source | Meaning |
+|---|---|
+| `scanner` | Default source |
+| `eval` | Set when an eval failure produced the finding; eval-sourced findings also carry an `eval_target_type` of `span`, `trace`, or `session` |
+
+## Event count & time fields
+
+Fields on an issue that count its events and place it in time.
+
+| Field | Counts |
+|---|---|
+| `total_events` | Every occurrence of the issue |
+| `unique_traces` | Distinct traces the occurrences fall across |
+| `unique_users` | Distinct users who hit the issue |
+| `first_seen` | Time of the issue's earliest occurrence |
+| `last_seen` | Time of the issue's most recent occurrence |
+
+## Traces tab columns & aggregates
+
+UI columns and summary cards on an issue's Traces tab.
+
+| Column |
+|---|
+| Trace ID |
+| Input |
+| Start Time |
+| Duration |
+| Tokens |
+| Cost |
+| Score |
+
+| Aggregate card |
+|---|
+| Total traces |
+| Avg score |
+| Avg turns |
+| P50 latency |
+| P95 latency |
+
+## List & query limits
+
+Values and bounds the feed and its tabs enforce: page sizes, default and maximum result counts, and the trends day window. A dash means that bound doesn't apply, only the minimum shown is enforced. The `limit` parameter appears twice below because it's bound differently on different surfaces of the app; the **Applies to** column says which surface each row's bounds belong to.
+
+| Parameter | Applies to | Min | Default | Max |
+|---|---|---|---|---|
+| `limit` | **Feed list** | 1 | 25 | 200 |
+| `offset` | **Feed list** | 0 | 0 | – |
+| `time_range_days` | **Feed list** | 1 | – | – |
+| `limit` | **Traces tab** | 1 | 50 | 500 |
+| `rep_limit` | **Overview** | 1 | 20 | 200 |
+| `days` | **Trends** | 1 | 14 | 90 |
+
+| Feed table page size (UI) |
+|---|
+| 10 |
+| 25 |
+| 50 |
+
+## Keep exploring
+
+
+
+ Filter, sort, and triage issues in the list view
+
+
+ The two independent axes every issue carries, and how they change
+
+
+ The fixed groups, categories, and fix layers behind every finding
+
+
diff --git a/src/pages/docs/error-feed/troubleshooting/analysis-does-not-finish.mdx b/src/pages/docs/error-feed/troubleshooting/analysis-does-not-finish.mdx
new file mode 100644
index 00000000..81464c6a
--- /dev/null
+++ b/src/pages/docs/error-feed/troubleshooting/analysis-does-not-finish.mdx
@@ -0,0 +1,46 @@
+---
+title: "Analysis doesn't finish"
+description: "Four causes for a stalled Fix tab run, matched to the message you see."
+---
+
+A cluster's [root cause analysis](/docs/error-feed/guides/run-root-cause-analysis) on the **Fix** tab can stall instead of landing a result. Match what you see against the causes below:
+
+- `Couldn't start the analysis. Please try again.` or `Couldn't connect to the server. Please try again.`: [The run never starts](#the-run-never-starts)
+- `Couldn't reach the investigator — the connection dropped. Hit Re-run.`: [Connection dropped after the run started](#connection-dropped-after-the-run-started)
+- Same message, but the workspace has no credit left: [Workspace out of credit](#workspace-out-of-credit)
+- No message, but the run's been going for an hour: [Run exceeded the one-hour cap](#run-exceeded-the-one-hour-cap)
+
+The credit is taken when the run starts and refunded if the run fails, so a failed run should net out to nothing.
+
+## The run never starts
+
+`Couldn't start the analysis. Please try again.` or `Couldn't connect to the server. Please try again.` The request to start the run failed outright. Nothing started. Re-run to try again.
+
+## Connection dropped after the run started
+
+`Couldn't reach the investigator — the connection dropped. Hit Re-run.` The run did start. The connection carrying its progress back died before anything came through. Re-running is usually safe, but the same message also shows up when the workspace has no credit left, and re-running there just spends another credit without landing a result. Check [Workspace out of credit](#workspace-out-of-credit) before you re-run again.
+
+## Run exceeded the one-hour cap
+
+Every run has a one-hour limit. If it's still going when that's reached, the run is cut off and no result lands in the thread. Re-run to start a fresh attempt.
+
+## Workspace out of credit
+
+This shows up as the same `Couldn't reach the investigator — the connection dropped. Hit Re-run.` message you'd see from a dropped connection, not as nothing happening. The workspace needs credit before a run can complete. If there's none left, the run fails with that message. Add credit to the workspace, then re-run.
+
+## What to do
+
+Before you re-run, confirm the cluster still has traces inside the [time range](/docs/error-feed/guides/triage-issues) you've got selected on the Feed. If the window has moved past everything in the cluster, widen or shift it so the cluster's traces fall inside, since re-running against an empty range spends a credit for nothing.
+
+Press **Re-run** in the cluster's header. Its tooltip reads `Re-run with current cluster state (1 credit)`, since each run, including a re-run, draws a fresh credit.
+
+## Dive deeper
+
+
+
+ Start a run, read the synthesis, and ask a follow-up
+
+
+ For when the Feed itself has nothing to analyze in the first place
+
+
diff --git a/src/pages/docs/error-feed/troubleshooting/issue-counts-look-wrong.mdx b/src/pages/docs/error-feed/troubleshooting/issue-counts-look-wrong.mdx
new file mode 100644
index 00000000..8c2bd9d3
--- /dev/null
+++ b/src/pages/docs/error-feed/troubleshooting/issue-counts-look-wrong.mdx
@@ -0,0 +1,70 @@
+---
+title: "Issue counts look wrong"
+description: "Why a populated Feed's numbers don't match what you expected"
+---
+
+The Feed has rows, nothing looks empty or filtered away, but a number on it doesn't match what you expected: a total that's lower than you'd guess, a row that's gone missing, or two columns that don't seem to agree with each other. Find your symptom below:
+
+- Total is lower than you expected: [Counts only cover what got scanned](#counts-only-cover-what-got-scanned)
+- Row you were watching has vanished: [Duplicate clusters get folded together](#duplicate-clusters-get-folded-together) or [The time range drops rows, but not their counts](#the-time-range-drops-rows-but-not-their-counts)
+- Events looks lower than the number of occurrences you know about: [The Events column is really a trace count](#the-events-column-is-really-a-trace-count)
+- Trend sparkline doesn't match the totals next to it: [The Trend sparkline runs on a fixed 14-day window](#the-trend-sparkline-runs-on-a-fixed-14-day-window)
+- Count still doesn't add up after accounting for sampling: [The list mixes scanner and eval issues](#the-list-mixes-scanner-and-eval-issues)
+
+Each row in the Feed is a cluster: one or more matching findings grouped into a single issue. See [Understanding Error Feed](/docs/error-feed/concepts/understanding-error-feed) for how clusters and categories form, and [Error taxonomy](/docs/error-feed/reference/error-taxonomy) for the fixed set a row's category comes from.
+
+## Counts only cover what got scanned
+
+Every project has a sampling rate between 0% and 100%, and it decides what fraction of traces are ever scanned in the first place. A count on the Feed only ever reflects scanned traces, never every trace that actually ran.
+
+That gap gets big fast at a low rate. At a sampling rate of 20%, roughly one trace in five gets scanned, so five traces that hit the exact same failure can turn into a single scanned occurrence. Read literally, that looks like the error happened once. It happened five times; only one of those times got sampled.
+
+Fix: if a count seems too low for how often you believe something is failing, that's the sampling rate doing its job, not a bug in the count. Raise the rate so more traces get scanned. That only affects traces scanned from that point on; it doesn't rescan what already ran, so counts you're currently looking at won't change. See [Turn on Error Feed](/docs/error-feed/guides/turn-on-error-feed) for where that control lives.
+
+## Duplicate clusters get folded together
+
+Two clusters get folded into one when they're in the same category, each is the other's closest match in both directions, and the distance between them is within the merge threshold. If cluster A's nearest neighbor is B, but B's nearest neighbor is something else, they don't fold. A mutual match in the same category that's still too far apart doesn't fold either; all three conditions have to hold together.
+
+When a fold happens, the cluster with the larger member count absorbs the other, and the absorbed one stops appearing in the Feed as its own row. Its occurrences don't disappear, they now count toward the cluster that absorbed it. So an issue you were watching yesterday can vanish from the list today, not because it resolved, but because it was the smaller, untriaged side of a mutual match and got absorbed into a bigger cluster in the same category. A cluster you've already triaged is protected from this: it survives the merge even if it would otherwise be the smaller side.
+
+Fix: if a row you expected is missing, look for a similar issue in the same category with a higher count than you remember. That's very likely where it went.
+
+## The Events column is really a trace count
+
+A cluster row carries several separate counts, and what shows up in the Feed table isn't a plain readout of them. The column labeled **Events** doesn't count events at all, it renders unique traces, the number of distinct traces the cluster matched.
+
+The **Users** column is a distinct end-user count, separate from Events.
+
+Fix: don't read Events as an occurrence count, it's the unique-trace count sitting under a misleading header. Total occurrences aren't shown as a column anywhere in the table.
+
+## The time range drops rows, but not their counts
+
+Changing the Feed's time range changes which clusters qualify for the list, not what their numbers say. A cluster only stays in the list when it was last seen within the selected range; narrow the range and clusters that fall outside it disappear from the table entirely. The Events and Users figures on a row that does survive aren't windowed to that range at all, they're lifetime values, so they don't shrink just because you picked a narrower range.
+
+Fix: if a row you expected is missing after narrowing the time range, that's the row falling outside the last-seen window, not its counts dropping to zero. Widen the range and the row reappears with the same lifetime totals it always had.
+
+## The Trend sparkline runs on a fixed 14-day window
+
+The Trend column doesn't follow the time range you've set for the rest of the table. It's labeled **Trend (14d)** and stays fixed at 14 days no matter what range you pick, so it won't line up with the Events or Users totals sitting next to it in the same row. Those totals are lifetime counts, not range-bound ones, so this isn't something you can tune away by adjusting the time range, the mismatch is permanent.
+
+Fix: read the sparkline as a separate signal, not a breakdown of the totals beside it.
+
+## The list mixes scanner and eval issues
+
+Issues on the Feed come from two different sources: the automatic scanner working through sampled traces, and evaluations. Both land in the same list and count toward the same totals unless you filter by source.
+
+Fix: if you're trying to reconcile a count against the sampling math in [Counts only cover what got scanned](#counts-only-cover-what-got-scanned) and it's not adding up, check whether some of the rows you're counting are eval-created rather than scanner-created. Source isn't a column in the Feed table, so you can't tell by looking at a row; filter the list to a single source instead. See [Issue fields & filters](/docs/error-feed/reference/issue-fields#source-values) for the source values and the filter that isolates them.
+
+## Dive deeper
+
+
+
+ For when the table has zero rows, not just numbers that look off
+
+
+ The mental model behind findings, clusters, and how issues form
+
+
+ Where the sampling rate lives and how to raise it
+
+
diff --git a/src/pages/docs/error-feed/troubleshooting/no-issues-in-the-feed.mdx b/src/pages/docs/error-feed/troubleshooting/no-issues-in-the-feed.mdx
new file mode 100644
index 00000000..f6297c66
--- /dev/null
+++ b/src/pages/docs/error-feed/troubleshooting/no-issues-in-the-feed.mdx
@@ -0,0 +1,75 @@
+---
+title: "No issues in the feed"
+description: "Five causes for a Feed with zero rows, in the order to check them."
+---
+
+The Feed loads, but the table is empty. Work out which situation you're in before you touch anything:
+
+1. Which empty-state message does the table show? [Filters are hiding the rows](#filters-are-hiding-the-rows) quotes both in full.
+ - The filtered message means the data may already be there, just filtered out, so skip straight to that section
+ - The unfiltered message means no issues have been found. Work through the causes below, in order
+2. Check the causes in order:
+ - [The workspace has no Error Feed license](#the-workspace-has-no-error-feed-license)
+ - [The sampling rate is still 0](#the-sampling-rate-is-still-0)
+ - [The traces are too new](#the-traces-are-too-new)
+ - [The traces came in through the collector](#the-traces-came-in-through-the-collector)
+ - [Filters are hiding the rows](#filters-are-hiding-the-rows)
+
+The list starts with the license check because it's the cheapest to rule out: the answer is visible on the page you're already looking at. If you've worked through all five causes and the Feed is still empty, confirm traces are reaching this project at all: see [No traces appearing](/docs/observe/troubleshooting/no-traces-appearing).
+
+## The workspace has no Error Feed license
+
+Without the Error Feed capability, the Feed page can't show a table at all: it shows an upgrade message instead, and a direct API call for feed data comes back with a 402.
+
+The message reads **"This feature requires an upgrade."**, with a reason code and a **Contact us to upgrade** button. If instead you see **"Couldn't verify feature access."** with a **Retry** button, that's a different problem: the check itself failed transiently, not a licensing block, so retry it.
+
+Fix: if you're looking at "This feature requires an upgrade.", this is your cause. Use the **Contact us to upgrade** button.
+
+## The sampling rate is still 0
+
+Error Feed ships with a project's [sampling rate](/docs/error-feed/guides/turn-on-error-feed) at 0, which disables scanning entirely. Nothing gets sampled, so nothing can ever reach the Feed. This is the shipped default, not something anyone had to break, so it's by far the most common reason the Feed is empty.
+
+Fix: raise the project's sampling rate above 0. See [Turn on Error Feed](/docs/error-feed/guides/turn-on-error-feed) for where that control lives and how to pick a rate.
+
+## The traces are too new
+
+Scanning is triggered per trace, not on a timer, and only once a trace's root span has completed. Even then, Error Feed waits about ten seconds before sampling and scanning it. A trace that finished moments ago hasn't necessarily been scanned yet.
+
+Fix: give it roughly ten seconds after the trace completes, then refresh the Feed.
+
+## The traces came in through the collector
+
+Traces that arrive through the collector don't trigger a scan on arrival. They wait for a periodic sweep instead, which adds its own grace period on top of the ten-second wait above. Check with whoever set up tracing for this project to see whether traces route through the collector.
+
+The sweep dispatches a scan task for every 15 pending traces, and each trace holds for a 60-second grace period before it's eligible. Because collector-routed traces are picked up by that sweep rather than one at a time, they show up in occasional bursts rather than the steady trickle you'd see from a trace that triggers its own scan.
+
+Fix: wait at least 60 seconds after the trace lands before assuming scanning isn't working, since that's the grace period each trace holds before it's even eligible for a sweep, and scanning still waits the same ten seconds after that. Expect issues to land in bursts rather than one at a time.
+
+## Filters are hiding the rows
+
+The table's empty state tells you which situation you're actually in.
+
+- If your filters exclude everything currently in the Feed, it shows **"No errors match your filters"** / **"Try adjusting your search or filter criteria."**
+- If there genuinely are no issues, it shows **"No errors - everything looks good!"** / **"Errors captured by Future AGI will appear here."**
+
+Fix: if you're looking at the first message, clear or widen your [filters](/docs/error-feed/guides/triage-issues). The **Clear** control:
+
+- resets project, status, severity, fix layer, and search
+- never resets the time range
+- only appears once one of project, status, severity, or fix layer is set
+
+So widen a narrow time range yourself. Neither message rules the time range out, since it isn't part of what the table checks: widen it before you go back through the causes above.
+
+## Dive deeper
+
+
+
+ Raise the sampling rate and get a project scanning for the first time
+
+
+ The mental model behind findings, clusters, and how issues form
+
+
+ For when the Feed has rows, but a number on it doesn't add up
+
+
diff --git a/src/pages/docs/evaluation/guides/advanced-usage.mdx b/src/pages/docs/evaluation/guides/advanced-usage.mdx
index 26132316..8ae4ff97 100644
--- a/src/pages/docs/evaluation/guides/advanced-usage.mdx
+++ b/src/pages/docs/evaluation/guides/advanced-usage.mdx
@@ -28,7 +28,7 @@ A toggle that lets the evaluator search the web while it judges, for verdicts th
## Connectors
-Connectors let the evaluator call your own tools mid-judgment, the same way it uses web search, so it can check a claim against your database, confirm an ID exists, or verify a rule in an internal service. A connector is a tool you expose over the Model Context Protocol (MCP) and register once on the [MCP Connectors](/docs/falcon-ai/features/mcp-connectors) page; this section assumes you have one registered.
+Connectors let the evaluator call your own tools mid-judgment, the same way it uses web search, so it can check a claim against your database, confirm an ID exists, or verify a rule in an internal service. A connector is a tool you expose over the Model Context Protocol (MCP) and register once on the [MCP Connectors](/docs/falcon-ai/concepts/mcp-connectors) page; this section assumes you have one registered.
*Attach a knowledge base, or create one if you have none yet*
diff --git a/src/pages/docs/evaluation/guides/running-evaluations.mdx b/src/pages/docs/evaluation/guides/running-evaluations.mdx
index 23ed4e06..29f7c5d0 100644
--- a/src/pages/docs/evaluation/guides/running-evaluations.mdx
+++ b/src/pages/docs/evaluation/guides/running-evaluations.mdx
@@ -30,9 +30,9 @@ In evaluation, **offline** and **online** describe the data, not your connection
| Surface | Where you run it | Guide |
|---|---|---|
-| Dataset and experiments | Offline, over every row of a dataset | [Run experiments](/docs/dataset/features/experiments) |
+| Dataset and experiments | Offline, over every row of a dataset | [Run experiments](/docs/dataset/guides/run-an-experiment) |
| Traces | Online, on live spans, traces, and sessions | [Set up evals in Observe](/docs/observe/guides/setup-evals) |
-| Simulation | Over simulated conversations | [Run a simulation](/docs/simulation/features/run-simulation) |
+| Simulation | Over simulated conversations | [Run a simulation](/docs/simulation/guides/run-voice-simulation) |
| SDK | Programmatic runs, with local Code Evals that need no API key | [Evaluation SDK](/docs/sdk/evals) |
| CI/CD | On every pull request, gating the merge on eval scores | [Evaluate in CI/CD](/docs/evaluation/guides/cicd) |
diff --git a/src/pages/docs/falcon-ai/concepts/mcp-connectors.mdx b/src/pages/docs/falcon-ai/concepts/mcp-connectors.mdx
new file mode 100644
index 00000000..16106e0c
--- /dev/null
+++ b/src/pages/docs/falcon-ai/concepts/mcp-connectors.mdx
@@ -0,0 +1,72 @@
+---
+title: "MCP Connectors"
+description: "Connect external MCP servers so their tools sit beside Falcon's own"
+---
+
+## An MCP connector links Falcon to an external tool server
+
+An **MCP connector** is a workspace's connection to an external server that speaks the [Model Context Protocol](https://modelcontextprotocol.io): a workspace-level object with a name, a server address, and everything Falcon has learned about that server since. Once that server is connected, its tools sit beside Falcon's own [platform tools](/docs/falcon-ai/concepts/understanding-falcon-ai) in the same conversation turn, so a single request can read an evaluation and open an issue in your tracker without you switching tools. [Skills](/docs/falcon-ai/concepts/skills) can reach for those same enabled tools too, alongside platform tools, when a workflow calls for them. Falcon's connector panel frames the idea plainly: "Add MCP connectors to give Falcon access to external services like GitHub, Slack, databases, and more."
+
+Two moments decide what Falcon can actually do with a connector: discovery, when Falcon asks the server what it offers, and enabling, when you choose which of those discovered tools it's allowed to call. For the steps to add a connector and authenticate it, see [Connect an MCP server](/docs/falcon-ai/guides/connect-mcp-server).
+
+ Turn
+ Enabled --> Turn`} />
+
+## Discovered tools are not the same as enabled tools
+
+This is the distinction that matters most. Discovery is Falcon asking the connected server what it can do: it queries the server's tool list, and a successful discovery turns every one of those tools on, so Falcon can call everything the server offered. Enabling is what happens afterward, and it only ever narrows that starting set down: you choose which discovered tools stay on and disable the rest. Falcon may only call the subset you've left enabled, and that subset can never grow past what discovery found.
+
+
+Only use connectors from developers you trust. Future AGI does not control which tools developers make available and cannot verify that they will work as intended or that they won't change.
+
+
+## The states you'll see
+
+A connector moves through four stages, but the app tracks them with only three status chips: "Connected", "Pending", or "Inactive". The chip is coarser than the stages, so several of the stages below share the same chip.
+
+- **Added.** The connector exists with a name and a server address. Falcon hasn't confirmed it can reach or use anything yet, so the chip reads "Pending"
+- **Authenticated.** Falcon has verified it can talk to the server, and the chip changes to "Connected". If verification fails instead, the error shows on the connector card in the Customize panel, so you can see it without having to reproduce it
+- **Tools discovered.** Falcon keeps the tool list it got back from the server, and the chip still reads "Connected"
+- **Tools enabled.** This is where you narrow the default set down to just the tools you want Falcon to use, and the chip still reads "Connected" here too
+
+A connector can also be turned off outright, at which point the chip reads "Inactive" regardless of what was discovered or enabled underneath it.
+
+To choose exactly which discovered tools stay enabled, see [Choose connector tools](/docs/falcon-ai/guides/choose-connector-tools).
+
+## Choosing how a connector authenticates
+
+Different servers expect different things from a client, so a connector's authentication is a choice, not a fixed requirement, and it's a property of the server you're connecting to, not something Falcon decides for you:
+
+- **None.** Some servers need no authentication at all
+- **API key or bearer token.** Some expect a credential you hold and hand to Falcon directly
+- **OAuth.** Some run a full sign-in, where you approve access in the provider's own window rather than typing a secret into Falcon
+
+Check the server's own documentation, or ask whoever runs it, to find out which one applies.
+
+## Why it matters
+
+Vetting the server, and narrowing its enabled tools down to just what you want Falcon to use, is on you.
+
+## Keep exploring
+
+
+
+ Add a connector and authenticate it
+
+
+ Narrow a connector's enabled tools down to just what you want Falcon to use
+
+
+ Build workflows that can call connector tools alongside platform tools
+
+
diff --git a/src/pages/docs/falcon-ai/concepts/skills.mdx b/src/pages/docs/falcon-ai/concepts/skills.mdx
new file mode 100644
index 00000000..3ac8b75b
--- /dev/null
+++ b/src/pages/docs/falcon-ai/concepts/skills.mdx
@@ -0,0 +1,72 @@
+---
+title: "Skills"
+description: "Reusable instructions that shape how Falcon works, shown with example trigger phrases."
+---
+
+## What a skill is
+
+A **skill** is a saved set of instructions plus example phrases for what it's for. Turning one on changes how Falcon approaches the request and which [tools](/docs/falcon-ai/concepts/understanding-falcon-ai) it reaches for, including anything connected through [MCP Connectors](/docs/falcon-ai/concepts/mcp-connectors), rather than just answering off a single prompt. A trigger phrase is an example of how someone might ask for this skill, recorded on the skill and shown on its card.
+
+ Contains
+ Skill -->|"visible everywhere"| BuiltIn["Built-in skill"]
+ Skill -->|"scoped to one workspace"| Custom["Custom skill"]
+ subgraph Activation["How a skill becomes active"]
+ Menu["Picked from Skills menu"]
+ SlashCmd["Typed as slash command"]
+ end
+ Skill --> Activation
+ Menu --> Active["Active skill for the conversation"]
+ SlashCmd --> Active`} />
+
+## Built-in and custom skills
+
+Falcon ships with a set of built-in skills. They show up in every workspace, and nobody can edit or delete them: a built-in skill's card in the Customize panel offers only Duplicate, with no edit or delete control. If a built-in skill is close to what you need, duplicate it there and adjust the copy instead of trying to change the original.
+
+Custom skills belong to a workspace. Anyone who creates one is creating it for the whole workspace, not just themselves, so every member of that workspace sees it and can trigger it. See [Create a skill](/docs/falcon-ai/guides/create-skill) for how to open the editor and build one from scratch.
+
+## Turning a skill on
+
+A skill becomes active for a conversation in one of two ways:
+
+- Pick it from the Skills menu in the Falcon AI header
+- Type its slash command at the start of a message
+
+For example, typing `/debug-traces` at the start of a message runs the Debug Traces skill directly; if it doesn't match a skill, Falcon treats the text as an ordinary message instead.
+
+## What ships built in
+
+Every workspace includes these built-in skills:
+
+- Analyze Costs
+- Analyze Trace Errors
+- Build a Dataset
+- Analyze Cluster
+- Compare Models
+- Debug Traces
+- Fix with Falcon
+- Localize Errors
+- Optimize Prompts
+- Run Evaluations
+
+## Why it matters
+
+Skills exist to make the same investigation come out the same shape every time, whether you run it or a teammate does, instead of everyone describing what they want from scratch. A skill is guidance, not a macro: you can add constraints, redirect the investigation, or ask follow-up questions after it's active, and Falcon incorporates them rather than running a fixed script to the end.
+
+## Keep exploring
+
+
+
+ The chat interface that skills run inside of
+
+
+ Open the skill editor and build a custom skill
+
+
diff --git a/src/pages/docs/falcon-ai/concepts/understanding-falcon-ai.mdx b/src/pages/docs/falcon-ai/concepts/understanding-falcon-ai.mdx
new file mode 100644
index 00000000..901b9bd4
--- /dev/null
+++ b/src/pages/docs/falcon-ai/concepts/understanding-falcon-ai.mdx
@@ -0,0 +1,64 @@
+---
+title: "Understanding Falcon AI"
+description: "How a Falcon AI conversation works, and what each turn draws on"
+---
+
+## A conversation is a thread, a turn is one exchange
+
+A **conversation** is one thread with Falcon inside a workspace. A **turn** is one thing you ask plus everything Falcon does to answer it: reading your message, deciding what to use, calling tools if it needs to, and writing a response. A conversation is a sequence of turns; everything below describes what happens inside one of them.
+
+## What a turn draws on
+
+Every turn draws on four things:
+
+- **Your message, and any file attached to it.** The question or instruction you typed, plus anything you uploaded alongside it
+- **The page you asked from.** Falcon knows what part of the platform you were looking at when you asked, the same way it knows which evaluation you mean when you ask about one without naming it
+- **The skill that's active.** A [skill](/docs/falcon-ai/concepts/skills) is a packaged, repeatable workflow, built-in or custom, that Falcon follows with the right tools already loaded. If one is running this turn, it's carried along with the turn, the same as your message or the page
+- **The tool set Falcon loads for that turn.** Tools are what let Falcon look things up and make changes on the platform, rather than just describe them; the pool includes any external tools you've connected through [MCP Connectors](/docs/falcon-ai/concepts/mcp-connectors) alongside the built-in platform ones, and Falcon loads only a working subset of it for each turn
+
+Falcon picks that subset by reading your request and settling on a working area, roughly forty tools brought forward for the turn out of the hundreds it could load. How specific your ask is decides how wide that area gets:
+
+- A vague ask keeps the working area broad and the tool set wide
+- A specific ask narrows both
+
+You can also point Falcon at a working area directly, either by opening the [context selector](/docs/falcon-ai/guides/chat-with-falcon-ai) in the chat input and picking one, or by naming the area in your message.
+
+ TURN["Turn"]
+ TURN --> MSG["Message + attached file"]
+ TURN --> PAGE["Page you asked from"]
+ TURN --> SKILL["Active skill"]
+ TURN --> TOOLS["Tool set for this turn"]
+ POOL["Full pool of loadable tools"] -->|"~40 brought forward"| TOOLS`} />
+
+## Watching a turn run
+
+While a turn runs, the answer streams in as it's written. Each tool Falcon calls shows up as its own card: it starts with **Running...**, and once the call finishes you can open it to see the **Parameters** it was given, the **Result**, and the **Full output** behind that result. If the turn creates or changes something on the platform, a completion card appears at the end with a link straight to it.
+
+
+As a conversation grows very long, Falcon automatically condenses the earlier part into a summary so the thread stays usable, while the most recent turns are kept exactly as written. Falcon still draws on what was discussed early on, but only as that summary rather than the original wording, so if an exact detail from far back in the thread matters, restate it rather than assume Falcon recalls it precisely.
+
+
+## Cost and pace
+
+Each turn costs one AI credit. You're also capped at ten messages a minute; go past it and Falcon shows an error in the conversation asking you to wait before sending more. If your organization's AI credit balance runs out, the turn is refused with an error in place of an answer. Credit balance is tracked on [Billing & Pricing](/docs/admin-settings/billing-pricing).
+
+## Why it matters
+
+The page you ask from, the skill that's active, and how specific your ask is all shape the tool set Falcon brings forward for a turn, and that shapes the answer you get back. That matters most in a long back-and-forth, where the message limit and credit cost add up turn by turn.
+
+## Keep exploring
+
+
+
+ Open the chat, ask questions, upload files, and follow responses
+
+
+ Use built-in workflows or create custom slash commands
+
+
+ Connect external tools like Linear, Slack, and GitHub
+
+
diff --git a/src/pages/docs/falcon-ai/features/chat.mdx b/src/pages/docs/falcon-ai/features/chat.mdx
deleted file mode 100644
index dbc02992..00000000
--- a/src/pages/docs/falcon-ai/features/chat.mdx
+++ /dev/null
@@ -1,150 +0,0 @@
----
-title: "Using Falcon AI: Chat, File Upload, and Tool Calls"
-description: "Open Falcon AI from any page, ask questions, upload files, and get streaming responses with tool calls and completion cards."
----
-
-## About
-
-Falcon AI runs as a chat interface inside the Future AGI dashboard. It can be opened as a sidebar from any page or as a full-page view for longer conversations. The sidebar stays open while you navigate between pages, so context is never lost. Conversations save automatically and can be resumed later.
-
-Falcon AI automatically detects what page you are on and uses it as context. Ask "why is this score low?" while viewing an evaluation, and it knows which evaluation you mean. It can also fetch content from URLs you paste, extract text from uploaded files, and stream responses with real-time tool execution.
-
----
-
-## Opening Falcon AI
-
-
-
- Press `Cmd+K` (Mac) or `Ctrl+K` (Windows/Linux) to open a sidebar overlay on the right side of the dashboard. It stays open as you navigate between pages.
-
- 
-
-
- Click **Falcon AI** in the navigation sidebar to open the full-page view at `/dashboard/falcon-ai`. A conversation history panel on the left lets you search, rename, and delete past conversations.
-
- 
-
-
-
----
-
-## Asking questions
-
-Type a question in the input area and press Enter. Falcon AI detects the domain of your request and loads the right tools automatically.
-
-
-
-To reference a different page than the one you are on, either navigate there first or specify it in your message:
-
-> "On the evaluations page, which model had the highest faithfulness score?"
-
----
-
-## Adding context
-
-Falcon AI detects page context automatically based on the current dashboard page. You can also attach entities manually by clicking **+ Add context** in the input area. Up to 5 entities can be attached at a time. Context chips appear above the input with an X to remove them.
-
----
-
-## Quick actions and slash commands
-
-On a new conversation, quick action buttons appear above the input: **Analyze with compass**, **Create custom views**, **Build a dataset**, **Create an evaluation**, **Run simulation for my agent**. They disappear after the first message.
-
-
-
-Type `/` at the start of a message to open the command picker. All active skills, both built-in and custom, appear as slash commands. Select one to run its workflow in the current conversation.
-
----
-
-## File uploads
-
-Click the attachment button or drag files into the input area. Falcon AI extracts text content and uses it as context for your question.
-
-| File type | What happens |
-|-----------|-------------|
-| **PDF** | Text is extracted from all pages |
-| **Excel / CSV** | Spreadsheet data is converted to text |
-| **Word (.docx)** | Document text is extracted |
-| **Images (PNG, JPG)** | Image is encoded and sent to the model for visual understanding |
-| **Text / Markdown / JSON** | Content is included directly |
-
-
- Maximum file size is 10 MB per upload.
-
-
----
-
-## URL fetching
-
-Paste a URL in your message and Falcon AI automatically fetches its content. This works with:
-
-- **Web pages**: HTML is cleaned and converted to text
-- **JSON APIs**: Response is formatted as a code block
-- **GitHub raw files**: Content is included as a code block
-- **Jupyter notebooks**: Code and markdown cells are extracted
-
-Up to 3 URLs are fetched in parallel, with a maximum of 50 KB of content per URL.
-
----
-
-## Following responses
-
-Responses stream token by token with Markdown formatting. When Falcon AI calls platform tools, collapsible cards show each step:
-
-- A **spinner** while the tool is running
-- A **checkmark** when it completes
-- A **warning icon** if it errors
-
-
-
-When a tool creates or modifies a platform entity, a **completion card** appears with a direct link to the result.
-
-Falcon AI can call multiple tools in parallel when they are independent, and chains them sequentially when one depends on another. A single turn can run up to 50 tool-call iterations.
-
----
-
-## Stopping a response
-
-Click the **Stop** button in the input area while Falcon AI is streaming. The current tool execution is cancelled and the response ends at whatever has been generated so far.
-
----
-
-## Conversation history
-
-
-
- Click the **history** (clock) button at the top of the sidebar to see past conversations.
-
-
- The left panel shows all conversations with search. Right-click a conversation to rename or delete it.
-
-
-
-
-
-Conversations persist across sessions. If you close the browser and come back, your full history is available.
-
----
-
-## Reconnection
-
-If your connection drops mid-response, Falcon AI automatically replays missed events when you reconnect so you see the complete response.
-
----
-
-## Rate limits
-
-Falcon AI allows 10 messages per 60 seconds per user. If you hit the limit, wait briefly before sending the next message.
-
----
-
-## Next Steps
-
-
-
- Use built-in workflows or create custom slash commands.
-
-
- Connect external tools like Linear, Slack, and GitHub.
-
-
diff --git a/src/pages/docs/falcon-ai/features/mcp-connectors.mdx b/src/pages/docs/falcon-ai/features/mcp-connectors.mdx
deleted file mode 100644
index c8ca3ee6..00000000
--- a/src/pages/docs/falcon-ai/features/mcp-connectors.mdx
+++ /dev/null
@@ -1,121 +0,0 @@
----
-title: "MCP Connectors: Connect External Tools to Falcon AI"
-description: "Connect external MCP servers to Falcon AI to use tools from services like Linear, Slack, GitHub, Sentry, and custom APIs."
----
-
-## About
-
-Falcon AI comes with built-in tools for the Future AGI platform, but many workflows involve external services: project trackers, communication tools, monitoring systems, and internal APIs. MCP Connectors extend Falcon AI by connecting it to any server that implements the [Model Context Protocol](https://modelcontextprotocol.io). Once connected, Falcon AI discovers the server's tools and can call them during conversations alongside built-in platform tools.
-
-This means tasks like "create a Linear ticket for this failing evaluation" or "post this cost report to Slack" happen inside Falcon AI without switching tools.
-
----
-
-## Examples
-
-- **Project management**: Connect Linear, Jira, or Asana to create and update issues from evaluation or trace analysis.
-- **Communication**: Connect Slack or email to share reports and alerts directly.
-- **Monitoring**: Connect Sentry or PagerDuty to pull error context into debugging conversations.
-- **Internal APIs**: Connect custom MCP servers that expose your organization's tools.
-
----
-
-## Adding a connector
-
-
-
- Open Falcon AI settings and go to the **Connectors** section. Click **Add Connector**.
-
- 
-
-
-
- Fill in the connector fields:
-
- | Field | Required | Description |
- |-------|----------|-------------|
- | **Name** | Yes | Display name for the connector (e.g., "Linear", "Sentry") |
- | **Server URL** | Yes | The MCP server endpoint URL |
- | **Transport** | Yes | `streamable_http` (default, recommended) or `sse` (Server-Sent Events) |
- | **Auth type** | Yes | How to authenticate with the server (see below) |
-
-
-
- Choose the authentication method that your MCP server requires:
-
- | Auth type | Fields | Description |
- |-----------|--------|-------------|
- | **None** | -- | No authentication required |
- | **API Key** | Header name, Header value | Sends a custom header with each request (e.g., `X-API-Key: your-key`) |
- | **Bearer Token** | Token | Sends `Authorization: Bearer ` with each request |
- | **OAuth 2.1** | Client ID, Client secret, Auth URL, Token URL, Scopes | Full OAuth flow with automatic token refresh |
-
-
- For OAuth connectors, Falcon AI handles the entire authorization flow. After saving the connector, click **Authenticate** to open the OAuth consent screen. Tokens are stored securely and refreshed automatically when they expire.
-
-
-
-
- Click **Test Connection** to verify that Falcon AI can reach the MCP server and authenticate successfully. If the test fails, the error message is displayed so you can debug the configuration.
-
-
-
- Click **Discover Tools** to query the MCP server for its available tools. Falcon AI reads the server's tool schema and displays the list with names, descriptions, and parameter definitions.
-
- The discovery result is cached. Re-run discovery if the server adds new tools.
-
-
-
- Not all discovered tools need to be active. Select which tools Falcon AI should have access to from the discovered list. Only enabled tools appear in conversations.
-
- This is useful when a server exposes many tools but you only need a subset, keeping Falcon AI's tool set focused and reducing context window usage.
-
-
-
----
-
-## Using connector tools in chat
-
-Once enabled, connector tools appear in Falcon AI conversations alongside built-in platform tools. Falcon AI decides when to use them based on your request. Tool names from connectors are prefixed with the connector name to avoid collisions (e.g., `linear_create_issue`).
-
-**Examples:**
-
-> "Create a Linear ticket for the faithfulness regression we found in the last evaluation run."
-
-> "Post a summary of today's error spikes to the #ml-alerts Slack channel."
-
-> "Check Sentry for any new issues related to the summarization service."
-
----
-
-## Transport options
-
-MCP Connectors support two transport protocols:
-
-| Transport | How it works | When to use |
-|-----------|-------------|-------------|
-| **Streamable HTTP** | Standard HTTP POST requests with JSON-RPC 2.0 payloads | Default. Works with most MCP servers. |
-| **SSE** (Server-Sent Events) | Long-lived HTTP connection with server-pushed events | Use when the server requires SSE transport or for streaming tool results. |
-
-Falcon AI automatically tries multiple endpoint paths (with and without `/mcp` suffix) to find the correct one for your server.
-
----
-
-## Managing connectors
-
-- **Edit**: Update any connector field from the Connectors settings page. Re-test and re-discover after changes.
-- **Delete**: Remove a connector and all its cached tool schemas. Tools from deleted connectors are immediately unavailable in conversations.
-- **Re-authenticate**: For OAuth connectors, click **Authenticate** again if the authorization has been revoked or if scopes need to change.
-
----
-
-## Next Steps
-
-
-
- Learn the basics of the chat interface.
-
-
- Create custom workflows that can use connector tools.
-
-
diff --git a/src/pages/docs/falcon-ai/features/skills.mdx b/src/pages/docs/falcon-ai/features/skills.mdx
deleted file mode 100644
index 6c708f5b..00000000
--- a/src/pages/docs/falcon-ai/features/skills.mdx
+++ /dev/null
@@ -1,133 +0,0 @@
----
-title: "Skill Builder: Custom Slash Commands for Falcon AI"
-description: "Use built-in skills for common workflows or create custom slash commands that package multi-step instructions for your team."
----
-
-## About
-
-The same analysis gets repeated across conversations and team members: checking regressions, generating cost reports, investigating error spikes. Skills package these workflows into reusable slash commands. Type `/` in the chat input, select a skill, and Falcon AI follows the packaged instructions with the right tools loaded.
-
-Falcon AI ships with six built-in skills for common workflows. You can also create custom skills scoped to your workspace.
-
----
-
-## Built-in skills
-
-These skills are available in every workspace and cannot be edited or deleted.
-
-### Build a Dataset
-
-Guides you through creating a dataset step by step. Falcon AI asks for a name, helps define columns, and walks you through adding rows, whether manually, from a file, or with synthetic generation.
-
-**Example**: `/build-a-dataset` → "I need a dataset of customer support tickets with columns for query, response, and sentiment."
-
----
-
-### Debug Traces
-
-Investigates traces with quantified analysis rather than vague summaries. Falcon AI reports specific error counts, latency percentile distributions, and recurring patterns across spans.
-
-**Example**: `/debug-traces` → "Why are we seeing timeout errors on the summarization endpoint?"
-
----
-
-### Compare Models
-
-Runs tradeoff analysis across multiple model variants. Falcon AI evaluates cost, quality, and latency side by side and highlights which model wins on each dimension.
-
-**Example**: `/compare-models` → "Compare GPT-4o and Claude Sonnet on our QA dataset for faithfulness and cost."
-
----
-
-### Run Evaluations
-
-Helps select the right evaluation template for your use case and explains results in context. Falcon AI picks templates based on your data type and walks through the scores.
-
-**Example**: `/run-evaluations` → "Evaluate the customer-support dataset for hallucination and toxicity."
-
----
-
-### Optimize Prompts
-
-Analyzes prompt versions and produces specific, actionable suggestions. Instead of generic advice, Falcon AI compares outputs across versions and points to what changed and why.
-
-**Example**: `/optimize-prompts` → "My summarization prompt is producing outputs that are too long. Help me tighten it."
-
----
-
-### Analyze Costs
-
-Produces cost breakdowns with exact dollar amounts and percentage savings opportunities. Falcon AI segments by model, project, and time period.
-
-**Example**: `/analyze-costs` → "Show me a cost breakdown for the last 30 days by model."
-
----
-
-## Custom skills
-
-Create skills specific to your team's workflows. Custom skills are scoped to the workspace and available to all workspace members.
-
-### Creating a skill
-
-
-
- In the Falcon AI chat input, click the **customize** button to open the skill picker. Click **Create Skill** to open the editor.
-
- 
-
-
-
- Fill in the skill fields:
-
- 
-
- | Field | Required | Description |
- |-------|----------|-------------|
- | **Name** | Yes | Display name shown in the command picker (e.g., "Weekly Cost Review") |
- | **Description** | Yes | Short description shown below the name in the command picker |
- | **Icon** | No | Icon displayed next to the skill name |
- | **Instructions** | Yes | The prompt that Falcon AI follows when the skill is triggered. Write these as direct instructions for the AI. |
- | **Trigger phrases** | No | Phrases that activate the skill automatically when typed in a message. Press Enter after each phrase. |
-
-
-
- Skill instructions work best when they are specific and structured. Include:
-
- - **What to do first**: Which tools to call and in what order
- - **How to present results**: Tables, comparisons, summaries
- - **What to ask the user**: If the skill needs input, tell Falcon AI to ask for it
-
- **Example instruction for a weekly review skill:**
-
- ```
- 1. Get evaluation scores for all datasets in this workspace from the last 7 days.
- 2. Compare each dataset's scores to the previous 7-day period.
- 3. Flag any metric that dropped by more than 5%.
- 4. Present results as a table with columns: Dataset, Metric, This Week, Last Week, Change.
- 5. If any regressions are found, suggest which traces to investigate.
- ```
-
-
-
- Type `/` in the chat input to open the command picker. Select your skill to run it. You can also type a message after selecting the skill to provide additional context.
-
- Skills also trigger automatically when a message matches one of the configured trigger phrases.
-
-
-
-### Editing and deleting skills
-
-Open the skill picker, click an existing custom skill to open the editor. Update any field and save, or click **Delete** to remove it. Built-in skills cannot be edited or deleted.
-
----
-
-## Next Steps
-
-
-
- Learn the basics of the chat interface.
-
-
- Extend Falcon AI with tools from external services.
-
-
diff --git a/src/pages/docs/falcon-ai/guides/chat-with-falcon-ai.mdx b/src/pages/docs/falcon-ai/guides/chat-with-falcon-ai.mdx
new file mode 100644
index 00000000..acf1bb8d
--- /dev/null
+++ b/src/pages/docs/falcon-ai/guides/chat-with-falcon-ai.mdx
@@ -0,0 +1,120 @@
+---
+title: "Chat with Falcon AI"
+description: "Open Falcon AI, ask a question, and attach a file."
+---
+
+Falcon AI is the AI copilot built into the Future AGI dashboard, reachable from any page you're on. Ask it about your workspace and it'll look up what it needs to answer.
+
+## Open Falcon AI
+
+Click **Falcon AI** in the dashboard's navigation sidebar. It opens the full-page view at `/dashboard/falcon-ai`, with your conversations listed down the left and the chat itself in the center.
+
+
+*The Falcon AI item in the dashboard nav opens the full-page view*
+
+
+*The full page adds a conversation list; the side panel doesn't*
+
+From anywhere else, press `⌘K` (Mac) or `Ctrl+K` (Windows/Linux), or click the floating button in the bottom-right corner of the page, tooltipped **Falcon AI (⌘K)**. Either opens a panel that slides in from the side, so you can keep the page underneath in view while you ask something.
+
+
+*The floating button and ⌘K both open the same side panel*
+
+Opening the full page while the side panel is open closes the panel.
+
+Every conversation is saved. See [Manage conversations](/docs/falcon-ai/guides/manage-conversations) for how to find, rename, or delete one.
+
+## Start a new conversation
+
+A fresh conversation opens with "How can I help?" and, under it, "Ask about your data, evaluations, experiments, traces, and more."
+
+Below that sit five quick-action chips:
+
+- Analyse my error feed
+- Create an Imagine view
+- Build a dataset
+- Create an evaluation
+- Run simulation for my agent
+
+Click one to prefill the input, then edit it before sending, or type your own question instead.
+
+
+*The five chips disappear once the conversation has its first message*
+
+## Ask a question and send it
+
+Type your question into the input at the bottom and click **Send**, or press Enter to send and Shift+Enter to add a newline. Send stays inactive until there's something to send, either typed text or [an attached file](#attach-a-file), so an empty input can't be submitted by mistake.
+
+
+*Send lights up once there's text or a file attached*
+
+## Point it at the right context
+
+Falcon AI defaults to Auto, reading the page you're on to work out what you're asking about. When Auto picks the wrong area, open the context selector and set it directly, before you send your question. The seven options are:
+
+- Auto
+- Datasets
+- Evaluations
+- Tracing
+- Experiments
+- Agents
+- Prompts
+
+
+*Switch off Auto when you're asking about something other than the page you're on*
+
+## Attach a file
+
+Falcon AI reads whatever you attach to help answer your question, alongside anything you type. Click the **+** button in the input area, labeled **More**, and choose **Attach files**, or drag a file onto the input and drop it. Falcon AI accepts:
+
+- Plain text, CSV, HTML, Markdown, and JSON
+- PDF
+- Excel (`.xlsx`) and Word (`.docx`)
+- PNG, JPEG, GIF, and WebP images
+
+Each file tops out at 10 MB, and the **Attach files** menu option reads "Uploading..." while the file goes up.
+
+
+*The plus button is labeled More; Attach files is the option that opens the file picker*
+
+
+A file over 10 MB is dropped with no message. If an attachment you tried to add never shows up above the input, that's why, so check its size and try a smaller file.
+
+
+## Follow the answer as it streams
+
+The response streams in as it's generated, and the input stays disabled until it finishes.
+
+If Falcon AI needs a tool along the way, a card appears in the conversation, starting at "Running...". Open it to see **Parameters**, what was sent, and **Result**, what came back, with **Full output** available when there's more to show. [Understanding Falcon AI](/docs/falcon-ai/concepts/understanding-falcon-ai) covers how it decides which tool to reach for.
+
+
+*A tool card opens to Parameters and Result; Full output expands longer results*
+
+If a turn is taking too long, click **Stop**. It cuts the response off where it is and keeps whatever has been written so far, rather than discarding it.
+
+
+*Stop ends the turn but leaves the partial answer in the conversation*
+
+## Copy or rate an answer
+
+Under each answer sit three actions: **Copy**, which copies the response to your clipboard, and **Good response** or **Bad response**, which record whether it was useful.
+
+
+*Copy, Good response, and Bad response sit under every answer*
+
+## Pace and cost
+
+Falcon AI is limited to ten messages a minute, and that window counts chat messages, response ratings, and Stop clicks together, so a burst of rapid follow-ups will hit that ceiling. Past it, the next send fails inline.
+
+Each turn costs one AI credit.
+
+## Dive deeper
+
+
+
+ Use built-in workflows or create custom slash commands
+
+
+ Connect external tools like Linear, Slack, and GitHub
+
+
diff --git a/src/pages/docs/falcon-ai/guides/choose-connector-tools.mdx b/src/pages/docs/falcon-ai/guides/choose-connector-tools.mdx
new file mode 100644
index 00000000..fcb7f171
--- /dev/null
+++ b/src/pages/docs/falcon-ai/guides/choose-connector-tools.mdx
@@ -0,0 +1,43 @@
+---
+title: "Choose connector tools"
+description: "Choose which of a connector's discovered tools stay allowed in Customize."
+---
+
+A connector that's finished discovery hands Falcon AI every tool it found, already allowed. Narrowing that down to what you actually want Falcon AI calling happens in a separate step. This guide covers the Customize panel, where you allow or deny a [connector's](/docs/falcon-ai/concepts/mcp-connectors) tools for the whole workspace, and the chat input's Connectors submenu, which is a shortcut for managing connectors rather than a separate control.
+
+This guide assumes you've already added a connector, authenticated it, and run discovery on it. If you haven't, see [Connect an MCP server](/docs/falcon-ai/guides/connect-mcp-server) first.
+
+## Open a connector's tool permissions
+
+From the [Falcon AI full page](/docs/falcon-ai/guides/chat-with-falcon-ai), open **Customize** in the left rail, the panel where you manage skills and connectors. Pick **Connectors**, then select the connector you want to configure.
+
+Its **Tool permissions** section is where access is actually decided, splitting what the server offers into two groups: **Interactive tools**, which take an action through the connected server such as creating, updating, or sending something, and **Read-only tools**, which only fetch or look up information.
+
+## Allow or deny individual tools
+
+To stop Falcon AI from calling a single tool, flip its toggle from **Allowed** to **Denied**. This doesn't touch the rest of its group, so you can deny the odd tool you don't want called while leaving the rest of Interactive tools or Read-only tools allowed. There's no group-level deny-all chip, so denying a whole group means flipping its tools one at a time.
+
+If you want a group back to fully allowed after denying part of it, click that group's **Always allow** chip to allow every tool in it again in one move.
+
+
+Falcon AI can only call a tool that's both discovered and allowed. If the connected server stops offering a tool you'd allowed, the change is rejected with an "Unknown tools" error until you re-run [Discover Tools](/docs/falcon-ai/guides/connect-mcp-server).
+
+
+## Before a connector has discovered anything
+
+Before a connector has discovered any tools, the Tool permissions section isn't there yet. In its place sits a single note: "No tools discovered yet. Tools will appear after the connector successfully connects." That's expected for a connector you've just added, or one whose discovery just hasn't finished yet.
+
+## Manage connectors from the chat input
+
+The chat input's **+** menu has a **Connectors** submenu listing every connector configured for the workspace along with its connected state, and a **Manage connectors** link that opens the connector settings page. If none are configured yet, it reads "No connectors configured" instead.
+
+## Dive deeper
+
+
+
+ Put newly enabled tools to work, then organize the conversations that use them
+
+
+ Build a skill that calls these same allowed tools
+
+
diff --git a/src/pages/docs/falcon-ai/guides/connect-mcp-server.mdx b/src/pages/docs/falcon-ai/guides/connect-mcp-server.mdx
new file mode 100644
index 00000000..72781870
--- /dev/null
+++ b/src/pages/docs/falcon-ai/guides/connect-mcp-server.mdx
@@ -0,0 +1,70 @@
+---
+title: "Connect an MCP server"
+description: "Give Falcon AI access to a new MCP server, whatever transport and authentication it expects."
+---
+
+This guide adds an MCP server to Falcon AI as a connector, so its tools become available for Falcon AI to call in chat once you turn them on (see [MCP Connectors](/docs/falcon-ai/concepts/mcp-connectors) for what a connector is). You name it, point it at the server, choose how Falcon AI talks to it and signs in, then check that the connection actually works before you rely on it in chat.
+
+Before you start, make sure you have access to Workspace Settings in your workspace, the MCP server's URL, and any credential the server requires, such as an API key or bearer token.
+
+## Open the Connectors page
+
+Connectors live on the **Falcon AI Connectors** page, under **Workspace Settings** in Settings. Open Settings, then click **Falcon AI Connectors** in the sidebar to go straight there. Click **Add Connector** to open the form.
+
+You can also get here from inside a conversation: **Manage connectors** in the chat input's **+** menu takes you to this same page, or use **Add custom connector** in Customize to add one without leaving the chat.
+
+## Name it and point it at the server
+
+Give the connector a **Name** and its **Server URL**, the address of the MCP server you're connecting to. Both fields are required to save.
+
+
+Connector names must be unique within the workspace. If another connector already uses the name you typed, saving is rejected until you pick a different one.
+
+
+## Choose the transport
+
+Set **Transport** to how Falcon AI should talk to the server. **Streamable HTTP** is the normal choice for a current MCP server. Pick **SSE (Legacy)** instead if you're connecting to an older server that still expects that transport.
+
+## Choose the authentication
+
+Set **Authentication** to whatever the server expects. The form reveals the fields that method needs:
+
+- **None**: skips authentication entirely
+- **API Key**: adds **Header Name** and **Header Value** fields
+- **Bearer Token**: adds a single **Bearer Token** field
+- **OAuth 2.0**: no extra fields, you sign in after saving
+
+Fill in whatever fields appear, then click **Create** to save the connector.
+
+## Sign in to an OAuth connector
+
+If you set **Authentication** to **OAuth 2.0**, saving the connector doesn't sign you in. Authentication happens as a separate step. Once saved, select the connector in the list on the left to open its detail pane, where you'll find **Re-authenticate**. That's the button for both your first sign-in and any later ones: click **Re-authenticate** (it reads **Authenticating...** while it runs), then approve access in the provider's own sign-in window. The window closes itself once you approve, handing control back to Falcon AI, and the result lands as **Authentication updated.** or **Authentication failed.** If it fails, confirm you approved access in the provider's window and try again.
+
+## Test the connection and discover its tools
+
+With the connector saved and, for OAuth, signed in, select the new connector in the list on the left to open its detail pane. This is where **Test Connection**, **Discover Tools**, **Re-authenticate**, **Edit**, and **Delete** all live. Click **Test Connection** to confirm Falcon AI can reach it. While it runs, the button reads **Testing...**; it resolves to **Connection test succeeded.** or **Connection test failed.**
+
+Once the connection is good, click **Discover Tools** (it reads **Discovering...** while it runs) to have the server report what it offers. A successful pass shows something like **Discovered 4 tools.** A failed one shows **Tool discovery failed.**
+
+## Narrow down the tools
+
+A successful discovery turns every tool it found on, so Falcon AI can already call all of them. The next step is cutting that set back to the ones you actually want, covered in [Choose connector tools](/docs/falcon-ai/guides/choose-connector-tools).
+
+## Edit or delete a connector
+
+Select the connector in the list on the left to open its detail pane, then click **Edit** to change its name, URL, transport, or authentication, and click **Save** to save your changes. Re-run **Test Connection** and **Discover Tools** afterward to make sure they still match. Click **Delete** to remove a connector you no longer need.
+
+
+Delete removes the connector immediately, there's no confirmation prompt.
+
+
+## Dive deeper
+
+
+
+ Turn on the discovered tools Falcon AI is allowed to call
+
+
+ Put connector tools to work once they're enabled
+
+
diff --git a/src/pages/docs/falcon-ai/guides/create-skill.mdx b/src/pages/docs/falcon-ai/guides/create-skill.mdx
new file mode 100644
index 00000000..cd1f6052
--- /dev/null
+++ b/src/pages/docs/falcon-ai/guides/create-skill.mdx
@@ -0,0 +1,70 @@
+---
+title: "Create a skill"
+description: "Build, save, and run a custom Falcon AI skill"
+---
+
+A skill packages a multi-step Falcon AI workflow into a slash command that anyone in your workspace can reuse, so instead of re-explaining a recurring analysis in every conversation, you type one command and Falcon AI runs the workflow behind it. This guide walks through building your own, from opening the editor to running what you save.
+
+## Open the skill editor
+
+Skills are built from Falcon AI's full page view; see [Chat with Falcon AI](/docs/falcon-ai/guides/chat-with-falcon-ai) if you haven't opened it yet. For what a skill is, see [Skills](/docs/falcon-ai/concepts/skills).
+
+Click **Customize** in the left rail of the Falcon AI full page to open the Customize panel, then pick **Skills** from its nav to see the skills available in your workspace, with a search field for finding one by name. Click **Create Skill** at the bottom of that list to open the editor for a brand new skill.
+
+
+Already mid-conversation? The Skills menu in the chat header also has a **Create Skill** entry, so you can build a skill without leaving the chat. It's a separate dialog rather than the Customize editor: it adds an **Icon** field, which takes an MDI icon name such as `mdi:bug` and sets the icon shown against the skill in the Skills list and picker, and its button reads **Create** on a new skill instead of **Save**.
+
+
+## Fill in the skill
+
+The Customize editor asks for four fields.
+
+Name is required and limited to 100 characters, and it's how the skill is labelled in the Skills list, so keep it short.
+
+Description is optional. It shows in the skill picker, so it's worth a line explaining what the skill does.
+
+Instructions is required. It's the prompt Falcon AI follows once the skill runs. Describe the workflow and reasoning, not a rigid script of commands to execute:
+
+- Workflow: "Compare this week's scores against last week's, and flag any metric that dropped by more than 5%."
+- Script: "Call the scores endpoint for this week, call it again for last week, then subtract."
+
+Trigger phrases need at least one entry before you can save; each one is an example of how someone might ask for this skill. Type a phrase and press Enter to add it to the list.
+
+## Example: a weekly eval regression review skill
+
+Here's a complete skill that compares [Evaluations](/docs/evaluation) scores week over week, filled in field by field:
+
+- **Name:** Weekly Eval Regression Review
+- **Description:** Comparing this week's evaluation scores against last week's and flagging anything that dropped
+- **Instructions:**
+
+ ```
+ Pull the evaluation scores for every dataset in this workspace from the last 7 days, then pull the same metrics for the 7 days before that as a baseline. Compare the two periods per dataset and per metric, and treat any metric that dropped by more than 5% as a regression worth flagging. Present the findings as a table with columns for Dataset, Metric, This Week, Last Week, and Change, with the biggest drops listed first. If nothing regressed, say so plainly instead of returning an empty table.
+ ```
+
+- **Trigger phrases:** "weekly eval review" and "check eval regressions"
+
+## Save and run it
+
+Click **Save** at the bottom of the editor to store the skill; it reads Saving... while it works.
+
+The skill you save gets its own slash command. Run it in any conversation by typing `/` in the chat input and picking the skill from the list that appears.
+
+
+In the Customize editor, leave Name blank and it says "Name is required". Leave Instructions blank and it says "Instructions are required". Skip trigger phrases and it says "At least one trigger phrase is required". A name already in use by another skill is rejected.
+
+
+## Edit, delete, and duplicate skills
+
+Custom skills can be edited and deleted. Open one again from the Skills list in Customize, change whatever field needs it, and save to update it. **Delete** removes it from the workspace entirely, and that option is available on any custom skill; skills that shipped with Falcon AI don't offer it. If a shipped skill covers most of what you need but not quite all of it, open it and click **Duplicate** instead of building from scratch. That gives you your own editable copy, ready to rename and adjust into the variant you actually want.
+
+## Dive deeper
+
+
+
+ Add a connector so Falcon AI can reach outside tools
+
+
+ Find, rename, and delete past conversations
+
+
diff --git a/src/pages/docs/falcon-ai/guides/manage-conversations.mdx b/src/pages/docs/falcon-ai/guides/manage-conversations.mdx
new file mode 100644
index 00000000..f0050a51
--- /dev/null
+++ b/src/pages/docs/falcon-ai/guides/manage-conversations.mdx
@@ -0,0 +1,44 @@
+---
+title: "Manage conversations"
+description: "Look up past chats from either surface, then rename or delete them from the full-page view."
+---
+
+Every conversation you have with Falcon AI is saved. You can find one from either surface Falcon AI runs on, its [full-page view or its side panel](/docs/falcon-ai/guides/chat-with-falcon-ai), and rename or delete it from the full page.
+
+## Find a conversation
+
+Where the list of past conversations shows up depends on which surface you're using. On the full-page view, it's the left rail, with a **Search chats...** field above it. In the side panel, click **Chat history** in the header to open the same list in a popover, with its own **Search chats...** field at the top. Either field filters the list down to the conversation you're after as you type.
+
+## Start a new chat
+
+Click **New chat** to start over. It's available in the left rail on the full page and in the side panel's header.
+
+## Rename or delete a conversation
+
+Renaming and deleting are only available on the full-page view. If you're in the side panel, open the full page first.
+
+Hover over a conversation in the left rail to reveal its row menu, a **⋯** icon at the row's right edge.
+
+Click it and choose **Rename** to type the new name in the **Rename conversation** prompt, or choose **Delete** to remove the conversation.
+
+
+*Rename and Delete share the same row menu*
+
+
+Delete has no confirmation step: the conversation disappears from your list as soon as you click it, so make sure it's the right row before you choose it. Deleting the conversation you currently have open drops you back to a new, empty chat.
+
+
+## Titles and dropped connections
+
+Falcon AI generates a conversation's title at the end of your first turn, so a thread still labeled "New conversation" hasn't gotten a reply yet. If your connection drops while Falcon AI is answering, it replays whatever you missed on its own once the connection comes back.
+
+## Dive deeper
+
+
+
+ Use built-in workflows or create custom slash commands
+
+
+ Connect external tools like Linear, Slack, and GitHub
+
+
diff --git a/src/pages/docs/quickstart/setup-mcp-server.mdx b/src/pages/docs/falcon-ai/guides/use-the-mcp-server.mdx
similarity index 81%
rename from src/pages/docs/quickstart/setup-mcp-server.mdx
rename to src/pages/docs/falcon-ai/guides/use-the-mcp-server.mdx
index fed183e7..1293a5c5 100644
--- a/src/pages/docs/quickstart/setup-mcp-server.mdx
+++ b/src/pages/docs/falcon-ai/guides/use-the-mcp-server.mdx
@@ -1,11 +1,11 @@
---
-title: "Setup MCP Server: Future AGI with Claude, Cursor, or VS Code"
-description: "Set up the Future AGI MCP Server to interact with the platform via natural language from Claude, Cursor, or VS Code using Model Context Protocol."
+title: "Use the MCP Server in your IDE"
+description: "Drive Future AGI from Claude, Cursor, or VS Code instead of the dashboard"
---
-import MCPIDETabs from '../../../components/MCPIDETabs.astro';
+import MCPIDETabs from '../../../../components/MCPIDETabs.astro';
-## About
+## What the MCP Server is
The **Future AGI MCP Server** lets you interact with the entire Future AGI platform through natural language, directly from your AI coding environment. Instead of switching between the dashboard and your editor, you can run evaluations, upload datasets, generate synthetic data, and apply protection rules just by describing what you want in tools like Claude, Cursor, or VS Code. It's built on the [Model Context Protocol](https://modelcontextprotocol.io/introduction), a standard that connects AI models to external tools and services.
@@ -44,7 +44,7 @@ https://api.futureagi.com/mcp
With **Future AGI's MCP Server**, you can use natural language to:
- **Run automatic evaluations**: evaluate batch and single inputs on various [evaluation](/docs/cookbook/quickstart/first-eval) metrics, both on local datapoints and large datasets
-- **Prototype and observe your agents**: add [observability](/docs/observe/quickstart), evaluations while [prototyping](/docs/prototype) and deploying agents into production
+- **Build and observe your agents**: add [observability](/docs/observe/quickstart), and [evaluations](/docs/evaluation) while you build and deploy agents into production
- **Manage datasets**: upload, evaluate, download [datasets](/docs/dataset) and find insights
- **Add protection rules**: apply toxicity detection, prompt injection protection, and other guardrails automatically
- **Generate synthetic data**: describe your dataset and objective to generate synthetic data
diff --git a/src/pages/docs/falcon-ai/index.mdx b/src/pages/docs/falcon-ai/index.mdx
index 399afdac..dc1b3d50 100644
--- a/src/pages/docs/falcon-ai/index.mdx
+++ b/src/pages/docs/falcon-ai/index.mdx
@@ -1,81 +1,52 @@
---
-title: "Falcon AI: AI Copilot for the Future AGI Dashboard"
-description: "An AI copilot embedded in the Future AGI dashboard that handles platform tasks, runs analysis, and answers questions through natural language."
+title: "Overview"
+description: "Where Falcon AI lives in the dashboard and how it differs from the MCP Server."
---
-## About
+## What is Falcon AI?
-Falcon AI is a copilot built into the Future AGI dashboard. It has access to over 300 platform tools and can work across datasets, evaluations, traces, experiments, prompts, and admin settings through natural language. It knows what page you are on, what entity you are looking at, and acts on that context directly.
+**Falcon AI** is an agent built into the Future AGI dashboard. Describe what you want in plain language and it works the platform for you, reading the page you're on, calling the right tools, and acting directly instead of pointing you to where to click. It works the platform with a built-in set of tools, which [skills](/docs/falcon-ai/concepts/skills) and [MCP Connectors](/docs/falcon-ai/concepts/mcp-connectors) shape and extend.
-{/* TODO: Add hero screenshot showing Falcon AI sidebar with a multi-step conversation */}
+## Where it shows up
-You describe a task, Falcon AI executes it. You ask a follow-up, it goes deeper. A single conversation can span multiple features: start from an evaluation regression, drill into the failing traces, inspect the dataset behind them, and compare against a different model.
+It's available on all plans. Falcon AI runs on two surfaces: a full page, which carries the conversation list and lets you rename or delete conversations, and a side panel, which keeps the page underneath in view.
----
-
-## What Falcon AI can do
-
-**Analyze.** Ask questions about your data and get quantified answers, not summaries.
-
-> "Which eval metrics dropped this week compared to last week?"
-> "What's the p95 latency for the summarization endpoint?"
-> "Show me a cost breakdown by model for the last 30 days."
-
-**Create.** Build platform entities without leaving the chat.
+- **Full page:** open the **Falcon AI** entry at the top of the left navigation, or land there automatically right after logging in
+- **Side panel:** from any other page in the dashboard, click the button in its bottom-right corner, or press `⌘K` (Mac) or `Ctrl+K` (Windows/Linux)
-> "Create a dataset called qa-golden with columns for query, expected_answer, and context."
-> "Run faithfulness and hallucination evals on the customer-support dataset."
-> "Set up an A/B experiment comparing GPT-4o and Claude Sonnet on the QA dataset."
+## What you can use it for
-**Debug.** Search traces, drill into spans, correlate across features.
+Falcon AI gets used for three kinds of work:
-> "Show me traces with timeout errors from the last 24 hours."
-> "Find traces where the model hallucinated and show me what context was retrieved."
+- **Analyze** what's already on the platform, for example "which evaluation scores dropped this week," and get an answer instead of a dashboard to go dig through yourself
+- **Create** things, like "build a dataset from these production [traces](/docs/tracing/concepts/traces)"
+- **Debug**, for example "why did this trace fail," by pointing Falcon AI at the record and asking what went wrong
-**Chain.** Work across features in a single conversation. Each follow-up builds on the previous result.
+## Falcon AI vs the MCP Server
-> "The faithfulness score on run 12 dropped. Show me the failing traces, then compare the prompts used in run 11 vs run 12."
+Falcon AI lives in the dashboard and knows the page you're on, so it can act on the record you already have open. The [MCP Server](/docs/falcon-ai/guides/use-the-mcp-server) lives in your IDE and knows your code instead, so it fits work that happens in a codebase rather than on the platform. Reach for Falcon AI when you're working in the dashboard, and the MCP Server when you're working in your editor.
----
-
-## Key capabilities
-
-| Capability | Details |
-|------------|---------|
-| **Page-aware context** | Automatically detects the current dashboard page and entity. Ask "why is this score low?" and it knows which evaluation you mean. |
-| **300+ tools** | Covers datasets, evaluations, traces, experiments, prompts, agents, simulations, cost analytics, and admin settings. |
-| **Multi-step execution** | Chains up to 50 tool calls per turn. Runs independent calls in parallel, sequential calls in order. |
-| **Skills** | Pre-built and custom slash commands that package multi-step workflows. Type `/` to access them. |
-| **File and URL input** | Upload PDFs, CSVs, images, or paste URLs. Falcon AI extracts content and uses it as context. |
-| **MCP Connectors** | Connect external services (Linear, Slack, GitHub, Sentry) so actions like "create a ticket for this regression" work in chat. |
-
----
+## Keep exploring
-## Falcon AI vs MCP Server
+Start with the mental model, then explore skills, connectors, and the guides as you need them.
-Future AGI has two AI interfaces for different contexts:
-
-| | Falcon AI | MCP Server |
-|--|-----------|------------|
-| **Where** | Inside the dashboard (browser) | Inside your IDE (Cursor, Claude Code, VS Code) |
-| **Who** | Platform users browsing the dashboard | Developers writing code |
-| **Context** | Knows what page is open, what entity is being viewed | Knows the codebase and files being edited |
-| **Output** | Rich rendering: charts, tables, completion cards | Text-only responses |
-
-Both share the same tool layer.
-
----
-
-## Next Steps
-
-
-
- Open the chat, ask questions, upload files, and follow responses.
+
+
+ The mental model behind a conversation, what a turn draws on and what it costs
+
+
+ Open Falcon AI from the nav, a shortcut, or the full page, then ask a question
+
+
+ Find conversations from either surface, then rename or delete them from the full page
+
+
+ The two kinds of skills, how each gets switched on, and what ships by default
-
- Use built-in workflows or create custom slash commands.
+
+ What changes once an external tool server is wired into a conversation
-
- Connect external tools like Linear, Slack, and GitHub.
+
+ Add a server as a connector and confirm Falcon AI can reach its tools
diff --git a/src/pages/docs/faq.mdx b/src/pages/docs/faq.mdx
index e1ccef7b..34a10e93 100644
--- a/src/pages/docs/faq.mdx
+++ b/src/pages/docs/faq.mdx
@@ -45,15 +45,15 @@ Use retrieval-specific evals like context_adherence, chunk_attribution, and reca
**How can I import data?**
-Data can be added manually, via file upload, SDK, or imported from Hugging Face. See [Create New Dataset](/docs/dataset/features/create).
+Data can be added manually, via file upload, SDK, or imported from Hugging Face. See [Create New Dataset](/docs/dataset/guides/create-a-dataset).
**What are dynamic columns?**
-Dynamic columns generate data automatically by running prompts, evaluations, API calls, or code against your dataset rows. See [Dynamic Columns](/docs/dataset/concept/dynamic-column).
+Dynamic columns generate data automatically by running prompts, evaluations, API calls, or code against your dataset rows. See [Dynamic Columns](/docs/dataset/concepts/static-and-dynamic-columns).
**Can I generate synthetic data?**
-Yes. Define a schema (columns, types, constraints) and the platform generates realistic rows. See [Synthetic Data](/docs/dataset/concept/synthetic-data).
+Yes. Define a schema (columns, types, constraints) and the platform generates realistic rows. See [Synthetic Data](/docs/dataset/concepts/synthetic-data).
---
@@ -65,11 +65,11 @@ Simulation lets you test voice and chat AI agents against simulated customers in
**How do I run a voice simulation?**
-Create an agent definition, scenarios, and personas, then run a test from the platform. See [Run Voice Simulation](/docs/simulation/features/run-simulation).
+Create an agent definition, scenarios, and personas, then run a test from the platform. See [Run Voice Simulation](/docs/simulation/guides/run-voice-simulation).
**Can I run chat simulations from code?**
-Yes, using the Python SDK. See [Chat Simulation Using SDK](/docs/simulation/features/simulation-using-sdk).
+Yes, using the Python SDK. See [Chat Simulation Using SDK](/docs/simulation/guides/run-chat-simulation).
---
@@ -81,7 +81,7 @@ Annotations are human labels applied to AI outputs (traces, spans, sessions, dat
**What's the difference between inline and queue-based annotations?**
-Inline annotations are quick, ad-hoc labels from detail views. Queue-based annotations use managed campaigns with assignment, progress tracking, and agreement metrics. See [Inline Annotations](/docs/annotations/features/inline).
+Inline annotations are quick, ad-hoc labels from detail views. Queue-based annotations use managed campaigns with assignment, progress tracking, and agreement metrics. See [Inline Annotations](/docs/annotations/guides/annotate-without-a-queue).
---
@@ -97,18 +97,6 @@ Every edit creates a new version. Assign labels (Production, Staging) to version
---
-## Prototype
-
-**What is Prototype?**
-
-Prototype is a pre-production testing environment. You run multiple versions of your application side by side and compare eval scores, cost, and latency. See [Prototype Overview](/docs/prototype).
-
-**How do I choose a winning version?**
-
-Use the Choose Winner flow to weight metrics and rank versions. See [Choose Winner](/docs/prototype/features/choose-winner).
-
----
-
## Optimization
**How does optimization work?**
@@ -117,7 +105,7 @@ Optimization takes a prompt, runs it against your data, scores the outputs with
**Can I optimize from the UI without code?**
-Yes. See [Using the Platform](/docs/optimization/features/using-platform).
+Yes. See [Using the Platform](/docs/optimization/guides/run-an-optimization).
---
@@ -141,7 +129,7 @@ Protect screens inputs and outputs in real time across four dimensions: Content
**Can I use Protect with text, images, and audio?**
-Yes. Protect works across all three modalities. See [Run Protect via SDK](/docs/protect/features/run-protect).
+Yes. Protect works across all three modalities. See [Run Protect via SDK](/docs/protect/guides/run-protect-from-the-sdk).
---
@@ -173,11 +161,11 @@ Error Feed automatically analyzes traces from your Observe projects, identifies
**How do I add documents to a Knowledge Base?**
-Upload files via the [UI](/docs/knowledge-base/features/ui) or programmatically via the [SDK](/docs/knowledge-base/features/sdk).
+Upload files via the [UI](/docs/knowledge-base/guides/create-knowledge-base) or programmatically via the [SDK](/docs/knowledge-base/guides/manage-with-the-sdk).
**What file types are supported?**
-PDF, DOCX, DOC, TXT, and RTF. Maximum 5MB per file. See [Understanding Knowledge Base](/docs/knowledge-base/concepts/concept).
+PDF, DOCX, DOC, TXT, and RTF. Maximum 5MB per file. See [Understanding Knowledge Base](/docs/knowledge-base/concepts/understanding-knowledge-base).
---
diff --git a/src/pages/docs/get-started/connect-no-code-agents.mdx b/src/pages/docs/get-started/connect-no-code-agents.mdx
index 334643ab..45b17b67 100644
--- a/src/pages/docs/get-started/connect-no-code-agents.mdx
+++ b/src/pages/docs/get-started/connect-no-code-agents.mdx
@@ -8,7 +8,7 @@ An agent definition tells Future AGI which agent you're testing and how to reach
Any voice agent can be simulated as long as it's reachable by a **phone number**. **Vapi** and **Retell** are natively supported, so you can sync the agent's name and prompt straight from them
-Building a **chat** agent? Chat simulations run through the **SDK**, not this form. See [Chat Simulation Using SDK](/docs/simulation/features/simulation-using-sdk)
+Building a **chat** agent? Chat simulations run through the **SDK**, not this form. See [Chat Simulation Using SDK](/docs/simulation/guides/run-chat-simulation)
Every definition is versioned, so once you save it you can run tests against a specific version, compare versions, or roll back
diff --git a/src/pages/docs/get-started/create-your-first-prompt.mdx b/src/pages/docs/get-started/create-your-first-prompt.mdx
index 3300c9b9..fd9f0e09 100644
--- a/src/pages/docs/get-started/create-your-first-prompt.mdx
+++ b/src/pages/docs/get-started/create-your-first-prompt.mdx
@@ -69,10 +69,10 @@ Not getting a response? Try checking these:
## Dive deeper
-
+
Test the prompt at scale across a dataset
-
+
Pull the prompt into your app and run it programmatically
diff --git a/src/pages/docs/index.mdx b/src/pages/docs/index.mdx
index 6acd06c9..7fb6500c 100644
--- a/src/pages/docs/index.mdx
+++ b/src/pages/docs/index.mdx
@@ -9,7 +9,7 @@ It's built for the whole team shipping AI (engineers, product managers, and doma
## The Learning Loop
-Every part of Future AGI feeds the next. You [**prototype**](/docs/prototype) and [**simulate**](/docs/simulation) an agent before launch, [**evaluate**](/docs/evaluation) its outputs against built-in and custom metrics, [**observe**](/docs/observe) real traffic once it's live, and [**optimize**](/docs/optimization) from what you learn, then the cycle repeats
+Every part of Future AGI feeds the next. You [**simulate**](/docs/simulation) an agent before launch, [**evaluate**](/docs/evaluation) its outputs against built-in and custom metrics, [**observe**](/docs/observe) real traffic once it's live, and [**optimize**](/docs/optimization) from what you learn, then the cycle repeats

@@ -22,7 +22,7 @@ Because every product shares the same **traces, datasets, and scores**, the work
Future AGI is organized into six broad areas:
- Build and refine: Prototype, Agent Playground, Prompt, and Dataset
+ Build and refine: Agent Playground, Prompt, and DatasetOne gateway for routing, caching, guardrails, and cost control across 100+ providersTest agents against synthetic users and scenarios before launchScore quality with built-in and custom metrics, guardrails, knowledge bases, and human review
@@ -54,5 +54,5 @@ The fastest way to see Future AGI is to get your data flowing:
- [Create your first prompt](/docs/get-started/create-your-first-prompt)
-**Using Cursor or Claude Code?** Install the Future AGI MCP server to bring the platform and docs straight into your editor. See [Set up the MCP server](/docs/quickstart/setup-mcp-server)
+**Using Cursor or Claude Code?** Install the Future AGI MCP server to bring the platform and docs straight into your editor. See [Set up the MCP server](/docs/falcon-ai/guides/use-the-mcp-server)
diff --git a/src/pages/docs/knowledge-base/concepts/concept.mdx b/src/pages/docs/knowledge-base/concepts/concept.mdx
deleted file mode 100644
index 888839c9..00000000
--- a/src/pages/docs/knowledge-base/concepts/concept.mdx
+++ /dev/null
@@ -1,51 +0,0 @@
----
-title: "Understanding Knowledge Base: Content Types and Processing"
-description: "Explains what a Knowledge Base is, what content types are supported, and how files are indexed and processed in Future AGI."
----
-
-## About
-
-A **Knowledge Base** is a store of your organization's content that Future AGI indexes and makes available across the platform. When you upload documents, the platform processes and indexes them so they can be used as grounding context for synthetic data generation and evaluations.
-
-## When to use
-
-- **Synthetic data generation**: You want generated examples to reflect your domain, terminology, and procedures instead of generic text. Selecting a KB when creating a synthetic dataset grounds the output in your actual content.
-- **Evaluation**: You want to check whether model outputs are factually consistent with your organization's knowledge. The KB supplies the reference context that hallucination detection and grounding evals compare against.
-
-## Supported Content Types
-
-You can upload the following file types to a Knowledge Base:
-
-| File Type | Extensions |
-|---|---|
-| Word documents | `.doc`, `.docx` |
-| PDF documents | `.pdf` |
-| Plain text | `.txt` |
-| Rich text | `.rtf` |
-
-Maximum file size is 5MB per file.
-
-Examples of content that works well in a KB:
-
-- Technical documentation and manuals
-- FAQs and troubleshooting guides
-- SOPs and process workflows
-- Training materials and HR policies
-- Legal documents and compliance information
-- Product descriptions and specifications
-
-## File Processing
-
-After you upload files, the platform processes them automatically. Each file goes through one of three states:
-
-| Status | Description |
-|---|---|
-| Successful | Content extracted and indexed for use |
-| Processing | File is being processed |
-| Failed | Processing failed. You'll be notified and the file won't be usable. |
-
-## Next Steps
-
-- [Create KB Using UI](/docs/knowledge-base/features/ui): Upload files through the dashboard
-- [Create KB Using SDK](/docs/knowledge-base/features/sdk): Upload and manage knowledge bases programmatically
-- [Synthetic Data](/docs/dataset/concept/synthetic-data): Learn how KB grounds synthetic data generation
diff --git a/src/pages/docs/knowledge-base/concepts/understanding-knowledge-base.mdx b/src/pages/docs/knowledge-base/concepts/understanding-knowledge-base.mdx
new file mode 100644
index 00000000..85f37a0c
--- /dev/null
+++ b/src/pages/docs/knowledge-base/concepts/understanding-knowledge-base.mdx
@@ -0,0 +1,56 @@
+---
+title: "Understanding Knowledge Base"
+description: "Ground synthetic data, agent evals, and simulations in your own documents"
+---
+
+## A knowledge base is one named container
+
+A **knowledge base** is a single named container that belongs to your organization, optionally scoped to one workspace. It holds two things: the documents you uploaded, and the indexed form of their text that the platform builds from them. Each document is its own record, with its own name and its own status, but they all live inside the one container you named when you created it.
+
+Indexing happens once, at upload. When you add a file, the platform reads it and extracts its text right away. It does not wait for something else to ask for that content first. That single pass is why the container carries a status of its own:
+
+- **Processing**: its files are being read
+- **Completed**: its files are usable
+- **Failed**: a file's extraction didn't work; the reason shows on that file's row
+
+## Three surfaces read it
+
+- **[Synthetic data generation](/docs/dataset/concepts/synthetic-data)** (generates dataset rows from a schema you define) points at a knowledge base by name so the rows it produces echo your domain instead of reading like generic text
+- **[Agent-type evaluations](/docs/evaluation/concepts/eval-types)** (evals authored as the Agent Evaluator type, which can reason over multiple turns and use tools) attach a knowledge base to give the eval a reference to check the agent's output against
+- **A [Simulation agent definition](/docs/simulation/concepts/agent-definitions)** (the config that governs how a simulated agent behaves) can attach one knowledge base of its own, giving the simulated agent something to draw on when it answers
+
+|"holds"| DOCS["Documents you uploaded"]
+ KB -->|"holds"| IDX["Indexed text"]
+ SDG["Synthetic data generation"] -->|"reads"| KB
+ EVAL["Agent-type evaluation"] -->|"reads"| KB
+ SIM["Simulation agent definition"] -->|"reads"| KB`} />
+
+## Not your agent's retrieval store
+
+Read it as reference material, not as your agent's live retrieval path. When your production agent answers a real request, it is not opening this container and searching it. A knowledge base is a document store that the three surfaces above consult when they run. There is no versioning either: a knowledge base holds whatever documents are in it right now, not snapshots of what it held before. If you're looking for a live index your agent queries during retrieval, that's the dataset's [Retrieval column](/docs/dataset/reference/dynamic-column-methods), which connects to Pinecone, Qdrant, or Weaviate, not to a knowledge base.
+
+## Why it matters
+
+Without a knowledge base behind it, a generator or an evaluator has nothing to ground itself in: ask for synthetic data on a topic you never described and it produces plausible wording that isn't yours. Point that generation, or an agent-type eval, at a knowledge base and it draws on your actual wording and your actual steps instead.
+
+## Limits and supported files
+
+- **Supported file types**: PDF, DOCX, TXT, and RTF only, anything else yields no extracted text
+- **Size cap**: 1 GB per knowledge base, across all its documents combined
+
+## Keep exploring
+
+
+
+ Upload documents and start indexing
+
+
+ Add or remove documents from an existing container
+
+
+ Create, update, and attach knowledge bases programmatically
+
+
diff --git a/src/pages/docs/knowledge-base/features/sdk.mdx b/src/pages/docs/knowledge-base/features/sdk.mdx
deleted file mode 100644
index 46549de4..00000000
--- a/src/pages/docs/knowledge-base/features/sdk.mdx
+++ /dev/null
@@ -1,117 +0,0 @@
----
-title: "Create a Knowledge Base Using the Future AGI Python SDK"
-description: "Create and manage Knowledge Bases programmatically with the Future AGI Python SDK: create, update, add or remove files, and delete KBs."
----
-
-## About
-
-The Knowledge Base SDK lets you create and manage Knowledge Bases from code using the Future AGI Python SDK. You install the `futureagi` package, authenticate with API credentials, then call methods to create a KB with a name and file paths (or a directory), add more files to an existing KB, remove files by name, or delete entire KBs. Supported file types are PDF, DOCX, TXT, and RTF.
-
-## When to use
-
-- **Automation**: Create or update KBs from scripts, pipelines, or scheduled jobs.
-- **Bulk ingestion**: Upload many files or point at a directory path instead of selecting files one by one in the UI.
-- **Larger files**: Use the SDK when file size or volume exceeds UI limits.
-- **Reproducibility**: Version and replay KB setup in code (e.g. in a repo or notebook).
-- **Integrations**: Embed KB creation/updates in your own tools or workflows.
-
----
-
-## How to
-
-
-
- Install the Future AGI Python package:
-
- ```bash
- pip install futureagi
- ```
-
-
-
- Create a `KnowledgeBase` client with your API key and secret (from the Future AGI platform). Optionally pass `fi_base_url` for a custom API base.
-
- ```python
- from fi.kb import KnowledgeBase
-
- client = KnowledgeBase(
- fi_api_key="YOUR_API_KEY",
- fi_secret_key="YOUR_SECRET_KEY"
- )
- ```
-
-
-
- Call `create_kb` with a name and either a list of file paths or a single directory path. The SDK uploads the files and returns the same client with the new KB cached in `client.kb` (id, name, files). Supported extensions: `pdf`, `docx`, `txt`, `rtf`.
-
- ```python
- client = client.create_kb(
- name="my-knowledge-base",
- file_paths=["path/to/file1.pdf", "path/to/file2.txt"]
- )
- print(f"Created KB: {client.kb.id} — {client.kb.name}")
- ```
-
- To use all files in a directory:
-
- ```python
- client = client.create_kb(
- name="my-knowledge-base",
- file_paths="path/to/docs_folder"
- )
- ```
-
-
-
- To add files or rename an existing KB, use `update_kb`. The first argument is the KB name (the SDK resolves it to the existing KB). You can pass `new_name` to rename and/or `file_paths` to add more files.
-
- ```python
- client.update_kb(
- kb_name="my-knowledge-base",
- file_paths=["path/to/extra_file.pdf"]
- )
- # Or rename and add files:
- client.update_kb(
- kb_name="my-knowledge-base",
- new_name="my-renamed-kb",
- file_paths=["path/to/extra_file.pdf"]
- )
- ```
-
-
-
- To remove specific documents, call `delete_files_from_kb` with the **file names** (as stored in the KB), not file IDs. You can pass `kb_name` if the client is not already targeting that KB.
-
- ```python
- client.delete_files_from_kb(
- file_names=["file1.pdf", "file2.txt"],
- kb_name="my-knowledge-base" # optional if client already has this KB
- )
- ```
-
-
-
- To delete one or more KBs, use `delete_kb` with either `kb_ids` or `kb_names`. The client can target the current cached KB if you don’t pass either.
-
- ```python
- client.delete_kb(kb_ids=[str(client.kb.id)])
- # Or by name:
- client.delete_kb(kb_names=["my-knowledge-base"])
- ```
-
-
-
-
- Wrap SDK calls in try/except and handle `fi.utils.errors.SDKException` (and optionally `InvalidAuthError`, rate limits) in production. Keep API credentials out of version control (e.g. use environment variables).
-
-
-## Next Steps
-
-
-
- Full SDK reference for the Knowledge Base module with all methods and parameters.
-
-
- Create and populate a Knowledge Base from the platform without code.
-
-
diff --git a/src/pages/docs/knowledge-base/features/ui.mdx b/src/pages/docs/knowledge-base/features/ui.mdx
deleted file mode 100644
index eb607043..00000000
--- a/src/pages/docs/knowledge-base/features/ui.mdx
+++ /dev/null
@@ -1,91 +0,0 @@
----
-title: "Create a Knowledge Base Using the Future AGI Platform UI"
-description: "Create and populate a Knowledge Base from the Future AGI platform: name it, upload documents, and wait for processing to finish."
----
-
-{/* ARCADE EMBED START */}
-
-
-
-{/* ARCADE EMBED END */}
-## About
-
-The Knowledge Base UI lets you create and populate a Knowledge Base directly on the Future AGI platform. Name your KB, upload documents (PDF, DOCX, TXT, RTF) via drag-and-drop or file picker, and the platform validates, uploads, and ingests them with per-file status (Successful, Processing, Failed) so you can track progress and retry failures. Once ingestion finishes, the KB is available for synthetic data generation and evaluations.
-
-## When to use
-
-- **Quick setup**: Create a KB and add documents in a few clicks without writing code.
-- **Small to medium documents**: Upload PDF, DOCX, TXT, RTF files within the UI file-size limits.
-- **Visibility**: See processing status per file and retry or fix failed uploads from the same screen.
-- **Team workflow**: Anyone with platform access can create or update KBs.
-
----
-
-## How to
-
-
-
- In the Future AGI dashboard, open the **Knowledge Base** tab from the left-hand navigation. Click **Create Knowledge Base** to start a new KB.
- 
-
-
-
- Give the KB a meaningful name (e.g. `SalesPlaybook_Q2`). If you leave the name empty, the system will assign a default such as `Knowledge Base - n`.
- 
-
-
-
- Add one or more files:
- 
-
- - **Supported formats:** `.pdf`, `.doc`, `.docx`, `.txt`, `.rtf` (and similar document types supported by the platform).
- - **File size:** Check the UI for the per-file limit (e.g. 5MB in the UI; larger files may require the SDK).
- - **Drag-and-drop** or use the file picker to select files from your machine.
-
-
- For files above the UI limit (e.g. up to 100MB), use the [SDK](/docs/knowledge-base/features/sdk) to create or update the KB; the UI may show a message or sample code to guide you.
-
-
-
-
- After upload, each file is processed. Status is shown per file:
- 
-
- - **Successful**: Content extracted and available in the KB.
- - **Processing**: File is still being ingested.
- - **Failed**: Upload or processing failed. Use retry or tooltips in the UI to fix or remove the file.
-
- The Knowledge Base is ready to use only after all files have completed successfully.
-
-
-
- Once processing is complete, the KB is ready. You can:
-
- - Use it for **synthetic data generation** (e.g. when creating or configuring a synthetic dataset).
- - Use it in **evaluations** (e.g. context grounding or hallucination detection) by referencing the KB in your eval or dataset setup.
- - **Add or remove files** later via the same KB detail view, or use the SDK for bulk updates.
-
-
-
----
-
-## Next Steps
-
-
-
- Automate creation and large file ingestion with the Python SDK.
-
-
- How KB fits into synthetic data and evaluations.
-
-
diff --git a/src/pages/docs/knowledge-base/guides/create-knowledge-base.mdx b/src/pages/docs/knowledge-base/guides/create-knowledge-base.mdx
new file mode 100644
index 00000000..c3a8902c
--- /dev/null
+++ b/src/pages/docs/knowledge-base/guides/create-knowledge-base.mdx
@@ -0,0 +1,67 @@
+---
+title: "Create a knowledge base"
+description: "Name a knowledge base, upload its first files, and submit"
+---
+
+This walkthrough builds one [knowledge base](/docs/knowledge-base/concepts/understanding-knowledge-base), `FAQ knowledge base`, and loads it with two files in a single trip through the create drawer: open it, name the knowledge base, add the files, and submit. What follows also covers the four things that actually stop this from going through: a knowledge base name already in use, an oversized file, two files with the same name in one upload, and a file the platform can't read.
+
+
+Viewer and Workspace Viewer roles can't create, update, or delete a knowledge base.
+
+
+## Open the create drawer
+
+Click **Knowledge base** in the sidebar, then click **Create Knowledge Base** on the list screen. A drawer titled **Create knowledge base** opens with two tabs, **Upload** and **Import from SDK**. Stay on **Upload** for this walkthrough; the SDK tab is covered in [Manage with the SDK](/docs/knowledge-base/guides/manage-with-the-sdk).
+
+
+Without an EE license, the create button is disabled and the list shows "Knowledge Base requires an EE license key".
+
+
+## Name it
+
+Type `FAQ knowledge base` into **Name**. Leave it blank and the knowledge base still gets created, just under an auto-generated name like `Knowledge Base - N`, so it's worth typing one you'll recognize later.
+
+A name already used elsewhere in your organization is refused: the platform checks this before anything is saved and rejects the whole submission if it collides, rather than renaming it for you.
+
+## Add the files
+
+Drag `FAQ.docx` and `Future AGI Customer FAQ.docx` into the dropzone that reads "Choose a file or drag & drop it here", or click **Browse files** and pick both. Each file has to be 5 MB or under, in PDF, DOCX, RTF, or TXT. Uploads also count toward 1 GB of total storage.
+
+A few things can keep a file, or the whole submission, from going through:
+
+- **A file over 5 MB** opens a dialog titled "Large files must be shared with SDK" instead of adding it, with "Cancel" and "Ok,got it" buttons; "Ok,got it" switches the drawer to the Import from SDK tab
+- **Two files with the same name in one upload** are refused before anything is saved, so a second `FAQ.docx` in the same batch won't quietly overwrite the first
+- **A password-protected or corrupted file, or one with no extractable text** (an image-only PDF, a blank DOCX, an empty or non-UTF-8 TXT, an RTF that yields no text) is rejected with its own message explaining why that specific file failed
+
+Once both are added the drawer reads "Files uploaded: 2/2". A number below the total means one of the staged files is in error.
+
+
+*Create stays greyed out until the first file lands in the dropzone*
+
+## Submit
+
+Click **Create**. You're taken straight to the new knowledge base's own page, where the files are still being processed.
+
+**Create** stays disabled until at least one file has been added, and again while any staged file is showing an error; remove that file and the button re-enables. If you close the drawer after typing a name or adding files, it asks "Are you sure you want to close? Your work will be lost" with a **Confirm** button, since nothing is saved until you submit.
+
+## Watch it process
+
+The knowledge base's page shows "Processing New Files" with "You've added 2 file(s). We're updating the knowledge base to reflect the new data." while `FAQ.docx` and `Future AGI Customer FAQ.docx` finish ingesting. Both the knowledge base list and this page's file table refresh on their own roughly every 10 seconds, so you can leave the page open and watch the status clear without reloading.
+
+## Dive deeper
+
+
+
+ Add files to an existing knowledge base, and remove the ones you don't want
+
+
+ Create and update a knowledge base programmatically, past the UI's file-size limit
+
+
diff --git a/src/pages/docs/knowledge-base/guides/manage-with-the-sdk.mdx b/src/pages/docs/knowledge-base/guides/manage-with-the-sdk.mdx
new file mode 100644
index 00000000..550f15af
--- /dev/null
+++ b/src/pages/docs/knowledge-base/guides/manage-with-the-sdk.mdx
@@ -0,0 +1,110 @@
+---
+title: "Manage with the SDK"
+description: "Create, update, and delete a knowledge base from Python"
+---
+
+The platform's upload dropzone caps a single file at 5 MB, which is small next to a real policy manual or a scanned contract. Reach for the SDK when:
+
+- A file is bigger than 5 MB
+- You want to point at a whole directory instead of picking files one by one
+- A knowledge base is something your own pipeline builds and refreshes on a schedule
+
+If you're creating a knowledge base from the UI and want a starting point, its create drawer has an **Import from SDK** tab that hands you a ready-made Python snippet for creating one; see [Create a knowledge base](/docs/knowledge-base/guides/create-knowledge-base) for that flow.
+
+## Install the SDK
+
+```bash
+pip install futureagi
+```
+
+## Set your credentials
+
+The client reads your API key and secret from the environment. Get `FI_API_KEY` and `FI_SECRET_KEY` from Settings > API Keys (see [API Keys](/docs/admin-settings/api-keys)), then set them before you run your script:
+
+```bash
+export FI_API_KEY="your-api-key"
+export FI_SECRET_KEY="your-secret-key"
+```
+
+## Initialize the client
+
+```python
+from fi.kb import KnowledgeBase
+
+kb = KnowledgeBase()
+```
+
+## Create a knowledge base
+
+Call `create_kb` with a name and `file_paths`. Pass a list of paths to upload specific files:
+
+```python
+kb.create_kb(
+ name="FAQ knowledge base",
+ file_paths=["docs/refund-policy.pdf", "docs/shipping-policy.txt"],
+)
+print("Created FAQ knowledge base")
+```
+
+Or pass a directory path to upload everything in it:
+
+```python
+kb.create_kb(
+ name="FAQ knowledge base",
+ file_paths="docs/support_policies/",
+)
+```
+
+`create_kb` returns the client itself (`self`), so calls can be chained. If the call returns without raising an exception, the knowledge base was created.
+
+These rules apply whichever path you use above, since they're enforced on the server, not by the dropzone:
+
+- Only PDF, DOCX, TXT, and RTF files are indexed
+- A knowledge base holds up to 1 GB total
+- The name has to be unique in your organization
+- A single call can't upload two files with the same name
+
+`create_kb` fails on any of the above. Fix the file, rename, or split the upload, then re-run.
+
+## Add a file
+
+`update_kb` adds files to an existing knowledge base. Pass the knowledge base's name and the new `file_paths`:
+
+```python
+kb.update_kb(
+ kb_name="FAQ knowledge base",
+ file_paths=["docs/warranty-policy.pdf"],
+)
+```
+
+## Remove a file by name
+
+`delete_files_from_kb` removes specific files. It takes the file names as they're stored in the knowledge base, not file IDs: that's the file's base name, not its local path. The `create_kb` call above uploads `docs/shipping-policy.txt`, and it's removed here by its base name, `shipping-policy.txt`:
+
+```python
+kb.delete_files_from_kb(
+ file_names=["shipping-policy.txt"],
+ kb_name="FAQ knowledge base",
+)
+```
+
+## Delete the knowledge base
+
+`delete_kb` removes one or more knowledge bases at once, by name or by ID:
+
+```python
+kb.delete_kb(kb_names="FAQ knowledge base")
+```
+
+For the by-ID form and its full parameters, see the [Knowledge Base SDK reference](/docs/sdk/knowledgebase).
+
+## Dive deeper
+
+
+
+ Full parameter and return details for every method
+
+
+ Turn a knowledge base into training and evaluation data
+
+
diff --git a/src/pages/docs/knowledge-base/guides/update-knowledge-base.mdx b/src/pages/docs/knowledge-base/guides/update-knowledge-base.mdx
new file mode 100644
index 00000000..f5a364d5
--- /dev/null
+++ b/src/pages/docs/knowledge-base/guides/update-knowledge-base.mdx
@@ -0,0 +1,57 @@
+---
+title: "Update a knowledge base"
+description: "Add documents, rename, remove files, or delete a knowledge base you already have running."
+---
+
+A knowledge base rarely stays static: policies get revised, new documents arrive, and old ones get pulled. Continuing with the `FAQ knowledge base` from [Create a knowledge base](/docs/knowledge-base/guides/create-knowledge-base), this guide covers the four ways you keep it current: adding more documents, renaming it, removing individual files, and deleting it altogether.
+
+Open `FAQ knowledge base` from the knowledge base list (`/dashboard/knowledge`) by clicking its row to reach its detail screen at `/dashboard/knowledge/:knowledgeId`. Adding documents, renaming, and removing files all happen from that detail screen; deleting the knowledge base itself happens from the list instead.
+
+
+*The detail screen carries every file with its processing status, and **Add docs** in the header*
+
+
+**Create Synthetic data** on the detail screen, which [grounds a synthetic dataset in this knowledge base](/docs/quickstart/generate-synthetic-data), stays disabled until the knowledge base's status reads **Completed** on that screen. Adding or removing files restarts reprocessing, so check the status shown on the detail screen after either change before trying to generate synthetic data.
+
+
+## Add more documents
+
+From the knowledge base detail screen, click **Add docs**. This opens the same drawer you used in Create a knowledge base, now titled **Add files** with the helper text "Add more files to update your knowledge base." Pick your files the same way, then click **Add** to submit.
+
+Everything you add still counts against the knowledge base's 1 GB total, and a file whose name already exists in `FAQ knowledge base` is refused at submission rather than added as a duplicate.
+
+## Rename the knowledge base
+
+Click the pencil icon button on the detail screen to open the **Edit Name** dialog. It has a single field labeled **Knowledge base** holding the current name. Change it and click **Save**, or **Cancel** to leave it as it is.
+
+## Remove files
+
+From the knowledge base detail screen, select one or more file rows. A bar reading `{n} Selected` appears with **Delete** and **Cancel** buttons. Click **Delete** on the selection bar, then confirm in the dialog titled `Delete {n} file(s)`, which asks whether you're sure you want to delete the selected file(s).
+
+Confirming switches the detail screen to an "Updating Knowledge Base" state while the change processes, then a toast confirms the result, for example `{n} files have been deleted.`
+
+Selecting all rows in the table only selects up to 20 at a time. A deleted file is gone; there's no way to bring it back. Clicking **Delete** without selecting anything just warns "No files selected."
+
+## Delete the knowledge base
+
+Deletion happens from the knowledge base list, not the detail screen. Select one or more knowledge bases there to bring up the same `{n} Selected` bar, and click **Delete**. Confirm in the dialog that follows, titled **Delete Knowledge Base** for one or **Delete Knowledge Bases** for several, asking "Are you sure you want to delete this knowledge base?" A toast then reads "Knowledge base deleted successfully."
+
+A deleted knowledge base is gone along with its files. You can delete one even while its files are still processing; doing so stops that work in progress rather than waiting for it to finish.
+
+## Dive deeper
+
+
+
+ Add, rename, and remove knowledge base content programmatically
+
+
+ Ground a synthetic dataset in a completed knowledge base
+
+
diff --git a/src/pages/docs/knowledge-base/index.mdx b/src/pages/docs/knowledge-base/index.mdx
index 581d8619..164577a3 100644
--- a/src/pages/docs/knowledge-base/index.mdx
+++ b/src/pages/docs/knowledge-base/index.mdx
@@ -1,45 +1,43 @@
---
-title: "Future AGI Knowledge Base: Store and Ground Source Content"
-description: "Store your organization’s content in Future AGI to ground synthetic data generation and evaluations in real source material."
+title: "Overview"
+description: "Where the platform reads your indexed documents, and where to go next."
---
-## About
+## What is Knowledge Base?
-A **Knowledge Base (KB)** is a store of your organization’s content: FAQs, documentation, SOPs, manuals, policies, and product specs. Future AGI indexes this content and makes it available across the platform. When you generate synthetic data or run evaluations, the platform pulls from your KB so outputs stay grounded in real source material instead of drifting into wrong terminology, invented procedures, or generic answers.
+A knowledge base is a named set of documents you upload and index once.
-## How Knowledge Base Connects to Other Features
+Where it's used: [synthetic data generation](/docs/dataset/concepts/synthetic-data), [agent-type evaluations](/docs/evaluation), and a [Simulation agent definition](/docs/simulation/concepts/agent-definitions).
-- **Synthetic data generation**: When creating a synthetic dataset, you can optionally select a KB. The generator uses your documents as context, producing examples that reflect your domain and terminology. [Learn more](/docs/dataset/concept/synthetic-data)
-- **Evaluation**: Run hallucination detection and grounding evals that compare model outputs against what your documents actually say. [Learn more](/docs/evaluation)
-- **Protect**: Use KB content as reference material for guardrails that check whether responses align with your organization’s knowledge. [Learn more](/docs/protect)
-
-## Getting Started
+## Dive deeper
+ How indexing works, why it isn't live retrieval, and the size and file-type limits
+
+
- What a KB is, what content types are supported, and how files are processed.
+ Name a knowledge base and upload your first documents
- Create and manage a KB from the platform with drag-and-drop or bulk file upload.
+ Add or remove documents, or delete a knowledge base entirely
- Create and update knowledge bases programmatically with the Python SDK.
+ Create, update, and delete knowledge bases from code, same as in the UI
-
-## Next Steps
-
-- [Generate Synthetic Data](/docs/quickstart/generate-synthetic-data): Use a KB to ground synthetic data generation
-- [Knowledge Base Cookbook](/docs/cookbook/quickstart/knowledge-base): Upload documents and query with the SDK
diff --git a/src/pages/docs/observe/features/voice.mdx b/src/pages/docs/observe/features/voice.mdx
index e526ce5c..a21b08d5 100644
--- a/src/pages/docs/observe/features/voice.mdx
+++ b/src/pages/docs/observe/features/voice.mdx
@@ -24,7 +24,7 @@ Voice agents are hard to debug. Conversations happen in real time, across multip
- **Evaluate voice conversations**: Run evals (quality, bias, adherence) on conversation spans from voice calls.
- **Alerts on voice metrics**: Set monitors on voice project metrics and get notified when something degrades.
- **Transcripts and recordings for debugging**: Access transcript and recording URLs from the trace view.
-- **Multiple voice providers**: Support for Vapi, Retell so you can monitor agents regardless of provider.
+- **Multiple voice providers**: Support for Vapi and Retell so you can monitor agents regardless of provider.
---
diff --git a/src/pages/docs/optimization/concepts/choosing-an-optimizer.mdx b/src/pages/docs/optimization/concepts/choosing-an-optimizer.mdx
new file mode 100644
index 00000000..392835cd
--- /dev/null
+++ b/src/pages/docs/optimization/concepts/choosing-an-optimizer.mdx
@@ -0,0 +1,74 @@
+---
+title: "Choosing an optimizer"
+description: "Match your optimization problem to the right optimizer algorithm."
+---
+
+## Six optimizers, one signal each
+
+What separates the six optimizers isn't their name, it's the signal each one reads to decide what to try next.
+
+ R1["Random variation on wording"]
+ ROOT --> R2["Score surface over examples and settings"]
+ ROOT --> R3["Textual feedback from failures"]
+ ROOT --> R4["Mutation plus critique-and-refine"]
+ ROOT --> R5["Evolutionary search across generations"]
+ R1 --> RS["Random Search"]
+ R2 --> BS["Bayesian Search"]
+ R3 --> PT["ProTeGi"]
+ R3 --> MP["Meta-Prompt"]
+ R4 --> PW["PromptWizard"]
+ R5 --> GP["GEPA"]`} />
+
+## Random variation: Random Search
+
+Random Search's only lever is how many wording variations it tries. It has no model of the score surface behind which variation to try next, so each [candidate prompt](/docs/optimization/concepts/understanding-optimization) is an unguided guess rather than a targeted edit. That's exactly why it's cheap: its budget is the smallest of the six.
+
+## Modeling the score surface: Bayesian Search
+
+Bayesian Search is steered by how many optimization trials it runs and how large a slice of your examples each trial can draw on. Instead of touching the instructional wording, it models how the score responds to which few-shot examples get included, how many, and under what settings, and searches that surface directly rather than guessing at edits.
+
+It costs more than Random Search's cheaper read, because each trial is a modeled choice over the example range rather than one flat guess, and Bayesian Search runs more trials by default than Random Search runs variations.
+
+## Reading failures as text: ProTeGi and Meta-Prompt
+
+ProTeGi and Meta-Prompt split off the same branch of the signal tree: both read textual feedback from the failures and use it to write targeted fixes, but they structure the search differently.
+
+ProTeGi keeps a **beam**, the set of candidate prompts carried forward and edited in parallel, alive across **rounds** (each round is one pass through the search loop). It computes textual gradients (descriptions of what's failing) from the errors, and edits every beam member from those gradients. ProTeGi pays for that breadth: because each beam member generates multiple candidates every round rather than one, its true cost runs well past a flat beam-times-rounds count.
+
+Meta-Prompt carries a single evolving prompt through more rounds than ProTeGi rather than maintaining parallel candidates, so its fix comes from depth of iteration on one line instead of breadth across a beam. It has no beam to multiply against, so its cost tracks its round count directly, trading ProTeGi's parallel breadth for depth on a single candidate.
+
+## Mutate, then critique and refine: PromptWizard
+
+PromptWizard mutates the prompt's wording directly, then critiques and refines the mutations that survive. Because mutation departs from the original wording entirely rather than patching specific failures, it's suited to prompts where the wording itself, not any one instruction inside it, has become the ceiling. It carries a narrower beam than ProTeGi, so it isn't paying for parallel candidates, but its mutate rounds and refine iterations per retained mutation still add up before scoring.
+
+## Evolutionary search against a budget: GEPA
+
+GEPA is steered by a single budget: how many **metric calls**, each one a candidate prompt scored against your dataset, it's allowed to spend. Rather than budgeting in rounds, trials, or beam size, it runs evolutionary search across generations and caps the search directly in metric calls, the widest single budget of the six. GEPA reads failures through a separate reflection model, distinct from the generator model the optimized prompt will actually run on.
+
+## Which optimizer fits your situation
+
+Start from your own situation, not the algorithm list. The bullets below run from a cheap first look to the widest, most expensive search, and the choice isn't final: you can rerun the same prompt with a different optimizer later.
+
+- If you're not sure yet, or just need a cheap read on how much room the prompt has before committing to anything heavier, use [Random Search](/docs/optimization/reference/optimizers/random-search)
+- If the wording already works but the few-shot examples feel arbitrary, use [Bayesian Search](/docs/optimization/reference/optimizers/bayesian-search)
+- If you already know where the prompt fails and want the algorithm to act on that feedback, use [ProTeGi](/docs/optimization/reference/optimizers/protegi) for several fixes explored in parallel, or [Meta-Prompt](/docs/optimization/reference/optimizers/meta-prompt) for fewer paths iterated longer
+- If targeted edits have stopped moving the score and the wording itself seems to be the ceiling, use [PromptWizard](/docs/optimization/reference/optimizers/promptwizard)
+- If there's budget for the widest search, use [GEPA](/docs/optimization/reference/optimizers/gepa)
+
+## Keep exploring
+
+
+
+ Walk a run's score graph, trial list, and per-row detail
+
+
+ Run optimization from code with the agent-opt library
+
+
+ Config keys, parameters, and defaults for each optimizer
+
+
diff --git a/src/pages/docs/optimization/concepts/concept.mdx b/src/pages/docs/optimization/concepts/concept.mdx
deleted file mode 100755
index 6a17589c..00000000
--- a/src/pages/docs/optimization/concepts/concept.mdx
+++ /dev/null
@@ -1,136 +0,0 @@
----
-title: "Understanding Prompt Optimization in Future AGI"
-description: "Explains how prompt optimization works in Future AGI: the feedback loop, key components, algorithms, and how to choose the right one."
----
-
-## About
-
-Prompt optimization is the process of iteratively improving a prompt using evaluation scores as feedback. You start with a baseline prompt, run it against your data, score the outputs, and let an algorithm generate better versions. Each round, the optimizer adjusts the prompt based on what scored well and what didn't.
-
-This is different from experimentation, which compares two or more fixed prompts side by side. Optimization takes one prompt and makes it better over multiple rounds.
-
-## How It Works
-
-The optimization loop has four components:
-
-1. **Dataset**: A set of input/output examples that the prompt runs against (e.g. questions and expected answers, articles and target summaries)
-2. **Generator**: The LLM that runs the prompt and produces outputs (e.g. GPT-4o-mini)
-3. **Evaluator**: Scores each output using an eval template (e.g. summary_quality, tone, groundedness)
-4. **Optimizer**: The algorithm that generates new prompt variations based on scores from previous rounds
-
-The process:
-
-```
-Baseline prompt
- ↓
-Run on dataset → Generate outputs
- ↓
-Score outputs with evaluator
- ↓
-Optimizer generates new prompt variations
- ↓
-Run variations on dataset → Score again
- ↓
-Repeat for N rounds
- ↓
-Return best prompt + score
-```
-
-Each round, the optimizer sees which prompts scored higher and uses that signal to generate the next set of candidates. After all rounds complete, you get the best prompt and its score.
-
-## Optimization vs Experimentation
-
-| | Optimization | Experimentation |
-|---|---|---|
-| **Goal** | Improve one prompt iteratively | Compare multiple fixed prompts |
-| **Process** | Algorithmic (automated rounds) | Manual (you define the variants) |
-| **Output** | Best prompt found + score | Side-by-side comparison of scores |
-| **When to use** | You have a prompt and want to make it better | You have multiple candidates and want to pick the best one |
-
-Typically you'd experiment first to find a promising prompt direction, then optimize that prompt to squeeze out more quality.
-
----
-
-## Choosing an Algorithm
-
-Future AGI supports 6 optimization algorithms. Use the tables below to pick one.
-
-### Quick Selection
-
-| Use case | Recommended optimizer | Why |
-|---|---|---|
-| Few-shot learning | Bayesian Search | Selects and formats examples intelligently |
-| Complex reasoning | Meta-Prompt | Deep failure analysis and full prompt rewrite |
-| Fixing specific errors | ProTeGi | Identifies and fixes failure patterns |
-| Creative / open-ended | PromptWizard | Diverse prompt exploration |
-| Production deployments | GEPA | Strong evolutionary search with good budgeting |
-| Quick baseline | Random Search | Fast, simple baseline |
-
-### Performance Comparison
-
-| Optimizer | Speed | Quality | Cost | Dataset size |
-|---|---|---|---|---|
-| Random Search | Fast | Basic | Low | 10-30 |
-| Bayesian Search | Medium | High | Medium | 15-50 |
-| Meta-Prompt | Medium | High | High | 20-40 |
-| ProTeGi | Slow | High | High | 20-50 |
-| PromptWizard | Slow | High | High | 15-40 |
-| GEPA | Slow | Excellent | Very High | 30-100 |
-
-### Decision Tree
-
-```
-Do you need production-grade optimization?
-├─ Yes → Use GEPA
-└─ No
- │
- Do you have few-shot examples in your dataset?
- ├─ Yes → Use Bayesian Search
- └─ No
- │
- Is your task reasoning-heavy or complex?
- ├─ Yes → Use Meta-Prompt
- └─ No
- │
- Do you have clear failure patterns to fix?
- ├─ Yes → Use ProTeGi
- └─ No
- │
- Do you want creative exploration?
- ├─ Yes → Use PromptWizard
- └─ No → Use Random Search (baseline)
-```
-
-For detailed parameters and configuration of each algorithm, see the individual algorithm pages linked in the sidebar.
-
----
-
-## Combining Optimizers
-
-You can run multiple optimizers sequentially for best results:
-
-```python
-# Stage 1: Quick exploration with Random Search
-random_result = random_optimizer.optimize(...)
-initial_prompts = [h.prompt for h in random_result.history[:3]]
-
-# Stage 2: Deep refinement with Meta-Prompt
-meta_result = meta_optimizer.optimize(
- initial_prompts=initial_prompts,
- ...
-)
-
-# Stage 3: Few-shot enhancement with Bayesian Search
-final_result = bayesian_optimizer.optimize(
- initial_prompts=[meta_result.best_generator.get_prompt_template()],
- ...
-)
-```
-
----
-
-## Next Steps
-
-- [Using the Python SDK](/docs/optimization/features/using-python-sdk): Run optimization programmatically
-- [Using the Platform](/docs/optimization/features/using-platform): Run optimization from the UI
-- [Using the Platform](/docs/optimization/features/using-platform): Run optimization from the UI
diff --git a/src/pages/docs/optimization/concepts/understanding-optimization.mdx b/src/pages/docs/optimization/concepts/understanding-optimization.mdx
new file mode 100644
index 00000000..a0590a43
--- /dev/null
+++ b/src/pages/docs/optimization/concepts/understanding-optimization.mdx
@@ -0,0 +1,80 @@
+---
+title: "Understanding optimization"
+description: "What a run fixes, what it produces, and how a trial gets ranked"
+---
+
+## Optimization finds a better-scoring prompt without hand-tuning
+
+Optimization takes a prompt you already have and the evals you use to score it, then searches for a version of that prompt that scores higher, without you hand-tuning the wording yourself. Most of what follows describes that process against dataset rows.
+
+## A run fixes its setup before anything runs
+
+An **optimization run** fixes four things at the start and doesn't change them while it's going:
+
+- A [prompt column](/docs/dataset/guides/run-a-prompt-on-every-row) to improve
+- An [optimizer algorithm](/docs/optimization/concepts/choosing-an-optimizer) that generates candidate prompts
+- A model that produces outputs
+- The [evals](/docs/evaluation/concepts/understanding-evaluation) you select as its objective
+
+From that fixed setup, a run produces trials. One is the baseline trial, and it holds your original prompt exactly as it stood before the run started. Every trial after it is a numbered variation trial, holding one new candidate prompt the optimizer decided to try next. How many variation trials a run produces is set when the run is configured; see [Run an optimization](/docs/optimization/guides/run-an-optimization).
+
+## A trial is one candidate prompt, scored row by row
+
+Every trial, baseline or variation, holds one candidate prompt and one result per dataset row. The baseline trial runs your original prompt against those rows, each variation trial its own candidate. That row-by-row result is where the actual measurement happens, not the prompt itself.
+
+Each row result carries a score and a reason for every eval you selected. Select three evals and a single row leaves three scores and three reasons behind it, one pair per eval.
+
+From there, each trial's row scores roll up into a single ranking:
+
+- The mean of a trial's row scores becomes that trial's average score
+- The average score ranks the trial against the other variation trials, not the baseline
+- The run reports the highest variation average score as the best score. The baseline's average score isn't a candidate in that ranking, it's the comparison line the variations are measured against, so a run can finish with a best score below the baseline
+
+A finished run keeps its winning candidate prompt alongside the baseline, so it's right there to review; see [Read optimization results](/docs/optimization/guides/read-optimization-results) for how to walk through it.
+
+Here's how everything above fits together:
+
+|"fixes"| Column["Prompt column"]
+ Run -->|"fixes"| Optimizer["Optimizer algorithm"]
+ Run -->|"fixes"| Model["Model"]
+ Run -->|"objective"| Evals["Selected evals"]
+ Run -->|"produces"| Baseline["Baseline trial"]
+ Run -->|"produces"| Variation["Variation trial"]
+ Baseline -->|"holds"| Row["Row result"]
+ Variation -->|"holds"| Row
+ Row -->|"per eval"| ScoreReason["Score and reason"]
+ ScoreReason -->|"mean"| Avg["Trial average score"]
+ Avg -->|"ranks (variations only)"| Best["Best score"]
+ Baseline -->|"average score"| BaselineScore["Baseline score"]
+ Best -->|"measured against"| BaselineScore`} />
+
+## What you watch while it runs
+
+A run moves through four steps in order: onboarding (initializing the run), running the baseline eval, starting trials, and finalizing the optimization. Its status is a coarser read on that same progress: Queue means the run is waiting to start, before onboarding begins. Running covers all four steps, from onboarding through finalizing. Completed means finalizing has finished. A run can also end Failed or Cancelled instead of Completed.
+
+You don't wait for Completed to see anything. Each trial becomes readable as soon as it finishes scoring, one at a time as the run works through them, rather than all arriving together at the end.
+
+## A run scores at most 50 rows, and evals decide what counts as better
+
+A run scores at most 50 dataset rows. Because the evals you pick are the objective, they're the entire definition of "better" for that run: change which evals are attached and the very same set of candidate prompts can rank in a different order.
+
+## Two surfaces, one engine: datasets and Simulation
+
+The same engine that optimizes a prompt column against dataset rows also optimizes an agent's prompt from a [Simulation](/docs/simulation) run. The optimizer algorithms, the trial structure, and the scoring shape described above carry over; the rest of the setup differs. The agent surface has no prompt column, since its run hangs off a test execution instead. Its status set drops Cancelled, and its fourth step is named Finalizing agent prompt rather than Finalizing optimization. The sample changes too: instead of dataset rows capped at 50, a Simulation run samples 5 to 10 scenarios from the test execution, or all of them when the execution has 10 or fewer.
+
+## Keep exploring
+
+
+
+ Configure and start a run from the UI
+
+
+ Walk a run's score graph, trial list, and per-row detail
+
+
+ Run optimization from code with the agent-opt library
+
+
diff --git a/src/pages/docs/optimization/features/using-platform.mdx b/src/pages/docs/optimization/features/using-platform.mdx
deleted file mode 100755
index 2e7bd853..00000000
--- a/src/pages/docs/optimization/features/using-platform.mdx
+++ /dev/null
@@ -1,95 +0,0 @@
----
-title: "Run Prompt Optimization from the Future AGI Platform UI"
-description: "Run prompt optimization from the Future AGI UI: pick a dataset and column, configure prompt and evals, run optimization, and apply the best prompt."
----
-
-## About
-
-**Using the platform** for optimization means running prompt optimization from the Future AGI web UI instead of code. You open a dataset, click **Optimize** to open the **Run Optimization** panel, then set the run name, the column that holds the prompt template, the optimizer (e.g. GEPA), the language model, and optimizer parameters (e.g. Max Metric Calls), and add evaluations. You click **Start Optimization** and the run executes on Future AGI’s backend; when it finishes, you review results in the **Optimization** tab, compare scores across variations, and apply the best prompt. No Python or SDK required. Everything is driven by the UI and your existing datasets and evals.
-
----
-
-## When to use
-
-- **No-code workflow**: Improve prompts without writing code; use the UI for configuration and runs.
-- **Dataset-centric**: Optimize a prompt that already lives in a dataset column; data and results stay in the platform.
-- **Team visibility**: Runs and results are stored in Future AGI so others can see and reuse them.
-- **Reuse existing evals**: Pick from preset or previously configured evaluations instead of defining them in code.
-- **Iterative refinement**: Run optimization, apply the best prompt, then run again if you want to refine further.
-
----
-
-## How to
-
-
-
- Go to the **Dataset** view and open a dataset that has the inputs and (if needed) model-generated outputs you use for optimization. In the top action bar, click **Optimize** (next to Run Prompt, Experiment, and Evaluate). Choose the **dataset column** that contains the prompt you want to improve.
- 
-
-
-
- In the **Run Optimization** panel, fill in:
- 
-
- - **Name**: Give the run a clear name (e.g. **GEPA-Feb27-1655**) so you can find it later in the **Optimization** tab.
- - **Choose Column**: Select the dataset column that contains the prompt template to optimize. The prompt column is used as the baseline for the run.
- - **Choose Optimizer**: Pick an optimizer (e.g. **GEPA**, Bayesian Search, Meta-Prompt, ProTeGi, Random Search, PromptWizard). Each has different trade-offs between speed and quality.
- - **Language Model**: Select the model used for optimization (inference and/or teacher model, depending on the optimizer).
-
-
-
- In the **Add Parameters** section, set optimizer-specific options. For example, **Max Metric Calls** limits the maximum number of metric evaluations; the suggested value is tuned for a good balance between speed and quality. Parameters vary by optimizer (e.g. num_rounds, beam_size). Use the recommended defaults unless you need to tune them.
- 
-
-
-
- Open the **Evaluations** section and select the evals to run on your dataset. Add and configure the evaluation metrics that will score each prompt variation. The optimizer uses these scores to rank variations and pick the best prompt.
- 
-
-
-
- When **Name**, **Choose Column**, **Choose Optimizer**, **Language Model**, parameters, and evaluations are set, click **Start Optimization**. The run executes on Future AGI’s backend; progress and results appear in the **Optimization** tab. Use **Cancel** to close without starting.
-
-
-
- After the run completes, open the **Optimization** tab for that dataset or run:
-
- - **Compare variations**: The system shows multiple optimized prompt versions ranked by evaluation scores.
- - **Check scores**: A table lists each prompt with its scores (e.g. Context Relevance, Context Similarity); the original prompt’s score is included for comparison.
- - **Pick the best**: Review the top variations; the best-performing prompt is highlighted. You can inspect each one before deciding.
-
-
-
- When you’ve chosen the best version:
-
- - **Apply** the optimized prompt so it replaces the original in your dataset or workflow.
- - **Export** the updated dataset if you need it elsewhere.
- - **Run another optimization** if you want to iterate further.
-
-
-
-
-Runs can be **paused and resumed**. Optimizer state is persisted after each trial, so you don’t lose progress if a run is interrupted.
-
-
----
-
-## Key Concepts
-
-- **Optimization run**: One run = one column (prompt template) + optimizer algorithm + evaluation templates + teacher/inference model. The run produces multiple trials.
-- **Baseline trial**: Your original prompt scored on the dataset. This is the starting point for comparison.
-- **Variation trials**: New prompts generated by the optimizer, each with an average score from your evals.
-- **Evaluation templates**: Define how each variation is scored (e.g. summary_quality, context_adherence). Use 1-3 that match your task; avoid conflicting criteria.
-
----
-
-## Next Steps
-
-
-
- Run optimization from code with the agent-opt library.
-
-
- Compare algorithms and choose the right one.
-
-
diff --git a/src/pages/docs/optimization/features/using-python-sdk.mdx b/src/pages/docs/optimization/features/using-python-sdk.mdx
deleted file mode 100755
index a8785fe1..00000000
--- a/src/pages/docs/optimization/features/using-python-sdk.mdx
+++ /dev/null
@@ -1,172 +0,0 @@
----
-title: "Run Prompt Optimization Using the Python SDK in Future AGI"
-description: "Run prompt optimization from code using the agent-opt Python library. Configure datasets, optimizers, and evaluation templates programmatically."
----
-
-## About
-
-**Using the Python SDK** means running prompt optimization programmatically with the `agent-opt` library (`pip install agent-opt`). You write Python that defines a dataset (list of dicts), an **Evaluator** (eval template + model for scoring), a **DataMapper** (dataset keys → eval inputs), and an **Optimizer** (e.g. Random Search, Meta-Prompt, ProTeGi, GEPA, Bayesian Search, PromptWizard). You call `optimizer.optimize(...)` and get back the best prompt and scores. The SDK gives you full control over which optimizer and parameters to use, so you can automate runs, plug into CI, or try multiple strategies in code. Unlike the platform UI, everything is in your script.
-
----
-
-## When to use
-
-- **Automation**: Run optimization from scripts or CI; no UI.
-- **Choice of optimizer**: Use Random Search, Bayesian, Meta-Prompt, ProTeGi, GEPA, or PromptWizard and tune their parameters in code.
-- **Custom data**: Keep your dataset in code (list of dicts) or load it from your own storage.
-- **Reproducibility**: Version your optimization config and dataset with your repo.
-- **Advanced config**: Set eval subset size, initial prompts, task description, and optimizer-specific options.
-
----
-
-## Core concepts
-
-The library is built around four components that work together:
-
-| Component | Role |
-| --- | --- |
-| **Optimizer** | Drives the improvement process. You pick one (e.g. `RandomSearchOptimizer`, `MetaPromptOptimizer`, `GEPAOptimizer`) based on your task. |
-| **Evaluator** | Scores prompt outputs using a specified eval template and model (e.g. Future AGI’s `turing_flash`). |
-| **DataMapper** | Maps your dataset fields to the keys the optimizer and evaluator expect (e.g. `input` → `article`, `output` → `generated_output`). |
-| **Dataset** | A list of dicts; each item is one example (e.g. `{"article": "...", "target_summary": "..."}`). |
-
----
-
-## How to
-
-
-
- Install the library and set environment variables so the evaluator can call Future AGI (and so your generator can call your LLM provider if needed).
-
- ```bash
- pip install agent-opt
- ```
-
- ```bash
- export FI_API_KEY="your_api_key"
- export FI_SECRET_KEY="your_secret_key"
- ```
-
- You can also pass `fi_api_key` and `fi_secret_key` into the `Evaluator` instead of using env vars.
-
-
-
- Build a list of dicts. Each dict is one example; keys should match what your prompt and DataMapper use (e.g. `article`, `target_summary` for summarization).
-
- ```python
- dataset = [
- {"article": "The James Webb Space Telescope has captured...", "target_summary": "The JWST has taken new pictures."},
- {"article": "Researchers have discovered a new enzyme...", "target_summary": "A new enzyme that rapidly breaks down plastics has been found."},
- # ... more rows
- ]
- ```
-
-
-
- The **Evaluator** scores each prompt’s outputs. The **DataMapper** maps your dataset keys to the eval’s expected keys (`input`, `output`, etc.).
-
- ```python
- from fi.opt.base.evaluator import Evaluator
- from fi.opt.datamappers import BasicDataMapper
-
- evaluator = Evaluator(
- eval_template="summary_quality",
- eval_model_name="turing_flash",
- fi_api_key="your_key", # or rely on env FI_API_KEY
- fi_secret_key="your_secret" # or rely on env FI_SECRET_KEY
- )
-
- data_mapper = BasicDataMapper(
- key_map={"input": "article", "output": "generated_output"}
- )
- ```
-
-
-
- Pick an optimizer that fits your task (e.g. **Random Search** for a quick baseline, **Meta-Prompt** for deep refinement, **GEPA** for production-grade results). Some optimizers need a generator or teacher model; others take model names and config.
-
- **Example: Random Search (simple baseline)**
-
- ```python
- from fi.opt.optimizers import RandomSearchOptimizer
- from fi.opt.generators import LiteLLMGenerator
-
- initial_generator = LiteLLMGenerator(
- model="gpt-4o-mini",
- prompt_template="Summarize this: {article}"
- )
-
- optimizer = RandomSearchOptimizer(
- generator=initial_generator,
- teacher_model="gpt-4o",
- num_variations=5
- )
- ```
-
- **Example: Meta-Prompt (deep reasoning)**
-
- ```python
- from fi.opt.optimizers import MetaPromptOptimizer
- from fi.opt.generators import LiteLLMGenerator
-
- teacher = LiteLLMGenerator(model="gpt-4o", prompt_template="{prompt}")
- optimizer = MetaPromptOptimizer(teacher_generator=teacher, num_rounds=5)
- ```
-
- For a detailed comparison, see the [Optimizers overview](/docs/optimization/concepts/concept).
-
-
-
- Call `optimizer.optimize()` with the evaluator, data mapper, dataset, and any optimizer-specific options (e.g. `initial_prompts`, `eval_subset_size`, `task_description`).
-
- ```python
- initial_prompt = "Summarize the following article: {article}"
-
- result = optimizer.optimize(
- evaluator=evaluator,
- data_mapper=data_mapper,
- dataset=dataset,
- initial_prompts=[initial_prompt],
- task_description="Generate a concise, one-sentence summary of the article.",
- eval_subset_size=10 # Use a subset of the data for faster evaluation per round
-)
- ```
-
-
-
- Use the returned `result` object: `result.final_score`, `result.best_generator.get_prompt_template()`, and `result.history` for each round or variation.
-
- ```python
- # Print the final score and the best prompt found
- print(f"Final Score: {result.final_score:.4f}")
- print(f"Best Prompt:\n{result.best_generator.get_prompt_template()}")
-
- # Review the history of the optimization
- for i, iteration in enumerate(result.history):
- print(f"\n--- Round {i+1} ---")
- print(f"Score: {iteration.average_score:.4f}")
- print(f"Prompt: {iteration.prompt}")
- ```
-
-
-
----
-
-
-
-## Next Steps
-
-
-
- Full SDK reference for the optimization module with all methods and parameters.
-
-
- Compare algorithms and choose the right one.
-
-
- Run optimization from the UI instead of code.
-
-
- Source code, advanced features, and contributing.
-
-
diff --git a/src/pages/docs/optimization/guides/optimize-from-the-sdk.mdx b/src/pages/docs/optimization/guides/optimize-from-the-sdk.mdx
new file mode 100644
index 00000000..bf69c8c4
--- /dev/null
+++ b/src/pages/docs/optimization/guides/optimize-from-the-sdk.mdx
@@ -0,0 +1,218 @@
+---
+title: "Optimize from the SDK"
+description: "Run prompt optimization from a Python script with agent-opt, for automation or working outside the platform UI."
+---
+
+This guide runs one optimization job from Python end to end: install `agent-opt`, set your Future AGI keys, build a dataset, configure an Evaluator and a BasicDataMapper, construct an optimizer, define a starting prompt, and read back the result.
+
+## Install and authenticate
+
+Install the library with `pip install agent-opt`. It calls the Future AGI platform to score prompts, so set `FI_API_KEY` and `FI_SECRET_KEY`, [your Future AGI API keys](/docs/admin-settings/api-keys). Your optimizer also calls an LLM to generate and refine prompts, through LiteLLM rather than the Future AGI platform, so set that provider's API key too (for example, `OPENAI_API_KEY` for an OpenAI model). Set all three as environment variables before you run anything:
+
+```bash
+pip install agent-opt
+export FI_API_KEY="your_api_key"
+export FI_SECRET_KEY="your_secret_key"
+export OPENAI_API_KEY="your_provider_key" # whichever provider your optimizer's LLM uses
+```
+
+You can also pass `fi_api_key` and `fi_secret_key` straight into the `Evaluator` you construct next, if you'd rather not rely on the environment.
+
+## Define the prompt
+
+This guide optimizes a one-sentence summarization prompt:
+
+```python
+summary_prompt = "Summarize the following article in one sentence: {article}"
+```
+
+## Build the dataset
+
+The dataset is a plain list of dicts. Each dict is one example the optimizer evaluates the prompt against. Every dict needs a key for each placeholder in your prompt template, since the generator fills the prompt straight from the row, so every row below needs an `article` key to match the `{article}` placeholder above:
+
+```python
+dataset = [
+ {
+ "article": "The James Webb Space Telescope has captured its clearest images yet of a distant exoplanet's atmosphere, revealing traces of carbon dioxide and methane.",
+ "target_summary": "JWST detected carbon dioxide and methane in a distant exoplanet's atmosphere.",
+ },
+ {
+ "article": "Researchers have discovered a new enzyme that breaks down PET plastic at room temperature, far faster than any previously known enzyme.",
+ "target_summary": "A newly discovered enzyme breaks down PET plastic much faster than before.",
+ },
+ # ... more rows
+]
+```
+
+The two rows above are enough to sanity-check the code path; a real run wants closer to dozens of rows, so the score the optimizer settles on reflects more than a couple of examples.
+
+`target_summary` above isn't a prompt placeholder; it's a reference value kept for your own comparison. Any keys like it are yours to keep.
+
+## Configure the Evaluator and the DataMapper
+
+The `Evaluator` scores every generated output, either with one of Future AGI's eval templates run against a chosen model (platform mode) or with your own local `metric` object. This guide uses platform mode, which takes `eval_template` and `eval_model_name`; it picks up `FI_API_KEY` and `FI_SECRET_KEY` from the environment you set earlier, so you don't need to pass them again here:
+
+```python
+from fi.opt.base.evaluator import Evaluator
+
+evaluator = Evaluator(
+ eval_template="summary_quality",
+ eval_model_name="turing_flash",
+)
+```
+
+`eval_template` is one of Future AGI's [built-in eval templates](/docs/evaluation/builtin) and `eval_model_name` is one of the [evaluator models](/docs/evaluation/concepts/evaluator-models) that can run it; `summary_quality` and `turing_flash` above are just this example's choices. For scoring with your own heuristic or LLM-judge code instead, construct `Evaluator` with a local `metric` object in place of `eval_template` and `eval_model_name`; see the [SDK reference](/docs/optimization/reference/sdk-api) for its full constructor.
+
+The `BasicDataMapper` connects your dataset's keys to the keys the eval template expects, through a `key_map` dict. Map the eval's `input` to whichever dataset field holds the source text, and map its `output` to the literal string `"generated_output"`, which the optimizer fills in with whatever the prompt produces at each iteration:
+
+```python
+from fi.opt.datamappers import BasicDataMapper
+
+data_mapper = BasicDataMapper(
+ key_map={"input": "article", "output": "generated_output"}
+)
+```
+
+The `key_map` above only maps `article` and the generated output, so `target_summary` isn't passed to the evaluator in this example.
+
+## Construct an optimizer
+
+Six optimizers ship with the library:
+
+- [Random Search](/docs/optimization/reference/optimizers/random-search)
+- [Bayesian Search](/docs/optimization/reference/optimizers/bayesian-search)
+- [Meta-Prompt](/docs/optimization/reference/optimizers/meta-prompt)
+- [ProTeGi](/docs/optimization/reference/optimizers/protegi)
+- [GEPA](/docs/optimization/reference/optimizers/gepa)
+- [PromptWizard](/docs/optimization/reference/optimizers/promptwizard)
+
+Each has its own constructor and its own reference page; see [Choosing an optimizer](/docs/optimization/concepts/choosing-an-optimizer) for how they compare. This guide continues with `GEPAOptimizer`, the widest and most expensive of the six searches; picking it is a matter of budget, not task type, so treat it as this example's choice rather than a summarization-specific recommendation. It takes a `reflection_model` for analyzing failures and a `generator_model` for producing the outputs being scored, both LiteLLM-routed models like the one mentioned in Install and authenticate above:
+
+```python
+from fi.opt.optimizers import GEPAOptimizer
+
+optimizer = GEPAOptimizer(
+ reflection_model="gpt-4-turbo",
+ generator_model="gpt-4o-mini",
+)
+```
+
+If you'd rather start with the simplest baseline, Random Search's reference page has the equivalent `optimize` call.
+
+## Run the optimization
+
+Every optimizer's `optimize` call takes the `evaluator`, `data_mapper`, and `dataset` you just built, plus arguments specific to that optimizer. GEPA also asks for `initial_prompts`, the `summary_prompt` you defined in Define the prompt above wrapped in a list, and `max_metric_calls`, a budget that caps the run at that many evaluator calls total. Other optimizers take other arguments in place of these; check the optimizer's own reference page for its exact call.
+
+```python
+result = optimizer.optimize(
+ evaluator=evaluator,
+ data_mapper=data_mapper,
+ dataset=dataset,
+ initial_prompts=[summary_prompt],
+ max_metric_calls=150,
+)
+```
+
+How long this takes depends on your dataset size, your model's latency, and `max_metric_calls`; start with a smaller budget while you're testing your setup, then raise it for a real run.
+
+### What can go wrong
+
+- A missing `FI_API_KEY` or `FI_SECRET_KEY` fails immediately when you construct the `Evaluator`, before `optimize` even starts
+- A missing or wrong provider key (`OPENAI_API_KEY` or whichever your model needs) doesn't stop generation or raise an error there: the generator swallows the exception and returns an empty string, which then gets scored normally, so the run keeps going while outputs come back empty and scores drop. That's the generator model only; a bad provider key for GEPA's `reflection_model` call is not caught the same way and does kill the run
+- A `key_map` that points to a field your dataset rows don't have is silently dropped, not an error; if scores look off, double-check that your `key_map` values match your dataset's actual keys
+
+## Read the result
+
+The returned `result` is an `OptimizationResult`:
+
+| Field | What it holds |
+|---|---|
+| `result.final_score` | The best score reached |
+| `result.best_generator.get_prompt_template()` | The winning prompt |
+| `result.history` | A list of entries, each with the `prompt` tried, its `average_score`, and the `individual_results` behind that score |
+
+`OptimizationResult` also carries `early_stopped`, `stop_reason`, `total_iterations`, and `total_evaluations`; see the [SDK reference](/docs/optimization/reference/sdk-api) for what each holds.
+
+Printing `result.final_score` and looping over `result.history` looks something like:
+
+```python
+print(f"Final score: {result.final_score:.4f}")
+
+for i, iteration in enumerate(result.history):
+ print(f"Round {i + 1}: {iteration.average_score:.4f}")
+```
+
+To use the winning prompt outside this script, take `result.best_generator.get_prompt_template()` and save it, or paste it directly into the application or platform prompt you optimized it for.
+
+## Full example
+
+This assumes the environment variables from Install and authenticate above are already exported.
+
+```python
+from fi.opt.base.evaluator import Evaluator
+from fi.opt.datamappers import BasicDataMapper
+from fi.opt.optimizers import GEPAOptimizer
+
+# 1. Dataset: each row is one example the optimizer scores the prompt against
+dataset = [
+ {
+ "article": "The James Webb Space Telescope has captured its clearest images yet of a distant exoplanet's atmosphere, revealing traces of carbon dioxide and methane.",
+ "target_summary": "JWST detected carbon dioxide and methane in a distant exoplanet's atmosphere.",
+ },
+ {
+ "article": "Researchers have discovered a new enzyme that breaks down PET plastic at room temperature, far faster than any previously known enzyme.",
+ "target_summary": "A newly discovered enzyme breaks down PET plastic much faster than before.",
+ },
+ # ... more rows
+]
+
+# 2. Prompt: the starting instruction GEPA will iteratively rewrite
+summary_prompt = "Summarize the following article in one sentence: {article}"
+
+# 3. Evaluator: scores each generated summary with the summary_quality template
+evaluator = Evaluator(
+ eval_template="summary_quality",
+ eval_model_name="turing_flash",
+)
+
+# 4. DataMapper: connects the dataset's keys to the eval's expected keys
+data_mapper = BasicDataMapper(
+ key_map={"input": "article", "output": "generated_output"}
+)
+
+# 5. Optimizer: GEPA evolves the prompt using a reflection model
+optimizer = GEPAOptimizer(
+ reflection_model="gpt-4-turbo",
+ generator_model="gpt-4o-mini",
+)
+
+# 6. Run
+result = optimizer.optimize(
+ evaluator=evaluator,
+ data_mapper=data_mapper,
+ dataset=dataset,
+ initial_prompts=[summary_prompt],
+ max_metric_calls=150,
+)
+
+# 7. Read the result
+print(f"Final score: {result.final_score:.4f}")
+print(f"Best prompt:\n{result.best_generator.get_prompt_template()}")
+
+for i, iteration in enumerate(result.history):
+ print(f"Round {i + 1}: {iteration.average_score:.4f}")
+```
+
+## Dive deeper
+
+
+
+ The full optimization module: every class, method, and return field
+
+
+ How the optimizers compare and when to reach for each one
+
+
+ Run the same kind of job from the UI instead of a script
+
+
diff --git a/src/pages/docs/optimization/guides/read-optimization-results.mdx b/src/pages/docs/optimization/guides/read-optimization-results.mdx
new file mode 100644
index 00000000..f8bd85ea
--- /dev/null
+++ b/src/pages/docs/optimization/guides/read-optimization-results.mdx
@@ -0,0 +1,75 @@
+---
+title: "Read optimization results"
+description: "Walk a run's score graph, trial list, and trial detail to see how a run performed."
+---
+
+Open a run from its dataset's **Optimization** tab once it's finished, and you'll land on the detail page this guide walks through. (Haven't started a run yet? See [Run an optimization](/docs/optimization/guides/run-an-optimization).)
+
+That page is where you find out whether the optimizer actually improved anything, both overall and on the individual evals you selected.
+
+## The run detail page
+
+Once a run is **Completed**, the detail page shows a score graph, a result bar sitting between the graph and the list with the improvement note and **View Column**, and the trial list itself.
+
+### The score graph
+
+The graph draws one line per eval, not one line per trial. The y-axis, labeled **Evaluation Score**, runs 0 to 100, and the x-axis has one category per trial: **Baseline**, **Trial 1**, **Trial 2**, and so on. Use the **Evaluations** multi-select above the graph to choose which eval lines are shown. The baseline is the run's starting point, and it's what every candidate is measured against.
+
+The pattern to look for is simple: the more a trial's lines pull above the baseline on the eval you care about, the more the run improved on it.
+
+Lines that stay bunched around the baseline mean the run plateaued and didn't find a meaningfully better prompt. If that happens, see [Choosing an optimizer](/docs/optimization/concepts/choosing-an-optimizer) to compare optimizers and try a different one for the next run.
+
+### The trial list
+
+The trial list is where you compare candidates. Each row is a trial the optimizer generated, labeled Trial 1, Trial 2, and so on, listed in the order they ran; the baseline has no row of its own here, even though it gets its own category on the graph. The list isn't sorted by score; instead, the strongest trial is flagged directly in the **Trial** column.
+
+
+*The strongest trial is called out without sorting the list*
+
+### While the run is going
+
+While a run is **Queue** or **Running**, the detail page shows a stepper for its four steps, onboarding, running the baseline eval, starting trials, and finalizing, and a 'Please wait while we complete the optimization...' loader instead of the graph, result bar, or trial list.
+
+A run that ends **Failed** shows an error in that same loader area. A run that ends **Cancelled** replaces the whole view with the **Optimization Stopped** panel and a **Re-Run Optimization** button instead, with no trial list at all; see [Restart a stopped run](/docs/optimization/guides/run-an-optimization#restart-a-stopped-run) for what that button does. The graph, result bar, and trial list described above only appear once a run reaches **Completed**.
+
+## Comparing evals
+
+Each trial's average score is the mean of its score across the rows the run scored, at most 50 rows from the dataset rather than the whole thing (see [A trial is one candidate prompt, scored row by row](/docs/optimization/concepts/understanding-optimization#a-trial-is-one-candidate-prompt-scored-row-by-row)).
+
+When a run has more than one eval, the average isn't the only number available. The trial list carries a column for each eval, with its score and change versus baseline, and the score graph draws a line for each eval too, so you can see which eval is driving a trial's average without leaving this page.
+
+
+A trial ranked lower on average can still be the right pick, if the eval you actually care about is the one it wins on.
+
+
+To look row by row, open a trial's **Trial Items** tab, covered below.
+
+## Open a trial
+
+Click into any trial in the list to see what's behind its score.
+
+### Prompt
+
+The **Prompt** tab shows the trial's full prompt on its own. Turn on **Show Diff** and it puts the baseline prompt and the trial's prompt side by side, with the changed lines highlighted, so you see exactly what the optimizer changed, added, or removed instead of spotting the differences yourself.
+
+
+*Show Diff highlights what the optimizer changed from the baseline prompt*
+
+### Trial Items
+
+The **Trial Items** tab is the row-by-row evidence behind the average. Each row is one dataset row the trial was scored against, showing the input, the output the model produced, and a score for each eval. If the run had more than one eval, this is where you see each eval's score for that specific row.
+
+## Show or hide columns
+
+If a run has many evals, the trial list can get wide. **View Column** on the result bar opens a menu to toggle which columns are shown, so you can hide the evals you don't need and focus the list on the ones you do.
+
+## Dive deeper
+
+
+
+ Fixes for runs that fail, stall, or don't improve
+
+
+ Run optimization from code with the agent-opt library
+
+
diff --git a/src/pages/docs/optimization/guides/run-an-optimization.mdx b/src/pages/docs/optimization/guides/run-an-optimization.mdx
new file mode 100644
index 00000000..f024911f
--- /dev/null
+++ b/src/pages/docs/optimization/guides/run-an-optimization.mdx
@@ -0,0 +1,68 @@
+---
+title: "Run an optimization"
+description: "Fill the Run Optimization drawer, launch a run, and get it stopped or restarted from the same tab."
+---
+
+Optimization runs live inside a dataset, next to the prompt column they're improving. A run writes new versions of that prompt, scores each against your evals, and returns the version that wins as a result you review; it doesn't overwrite the prompt in your dataset column. This guide walks through starting a run on a dataset with a `summary_prompt` column, using GEPA as the optimizer and `summary_quality` as the scoring eval, and covers stopping or restarting it afterward.
+
+
+You need a dataset with a column that Run Prompt created, since Choose Column only lists those, and at least one eval that scores that column's output. See [Running Evaluations](/docs/evaluation/guides/running-evaluations) for where to set one up if you don't have one yet.
+
+
+## Open the run drawer
+
+Open the dataset that holds the prompt column you want to improve, then go to its **Optimization** tab. The button is **Run Optimization** on an empty tab and **Optimize Prompts** in the grid header once runs exist; either one opens the same **Run Optimization** drawer.
+
+If the dataset doesn't yet have a column of generated outputs to optimize, the drawer shows a **Run Prompt** button in place of the fields below. Click it, or see [Run Prompt](/docs/dataset/guides/run-a-prompt-on-every-row), then reopen the drawer.
+
+## Fill the run drawer
+
+For this walkthrough, fill in the fields as follows. The first four are fixed fields in the drawer; the rest are parameter fields that change with the selected optimizer, and Evaluations is a separate accordion below them:
+
+- **Name**: leave the auto-generated column-optimizer-timestamp value, or edit it to something more recognizable; your edit is kept instead of the generated value
+- **Choose Column**: `summary_prompt`, the column holding the prompt to optimize
+- **Choose Optimizer**: GEPA, already selected by default; leave it as is for this walkthrough
+- **Language Model**: any available model in the list; any of them works here
+- **Optimization Objective**: `Produce concise, accurate summaries that capture the key points of the source text`, a goal statement describing what the optimized prompt should achieve
+- **Max Metric Calls**: 40, the suggested default; this is the total number of metric evaluations the run can spend
+- **Evaluations**: an accordion, not a field you choose from; picking `summary_prompt` loads whatever evals are already attached to that column, and every one of them scores the run. If none are attached, the accordion shows 'No evaluations added' with an **Add Evaluations** button. For this walkthrough, `summary_quality` is already attached to `summary_prompt` and loads in with it
+
+
+*The Run Optimization drawer with GEPA selected, showing Optimization Objective and Max Metric Calls*
+
+Optimization Objective is shared across all six optimizers; the remaining parameter fields change with whichever optimizer is currently selected. See [Optimizers](/docs/optimization/reference/optimizers) for the full field list by optimizer.
+
+An eval is the signal the optimizer improves against: it scores each candidate prompt. See [Understanding Evaluation](/docs/evaluation/concepts/understanding-evaluation) for how evals work. The run needs at least one before it will start; submitting without one is blocked with 'Add evaluations before starting your optimization run'.
+
+
+Closing the drawer partway through prompts a confirmation, 'Are you sure you want to close? Your work will be lost', so anything you've filled in is gone once you confirm.
+
+
+## Start the run
+
+Click **Start Optimization**. A successful submission shows 'Optimization created successfully' and takes you straight into the new run's page instead of leaving you on the run list. If it fails, a toast reads 'Failed to create optimization' when the server doesn't return a more specific error message; click **Start Optimization** again to retry.
+
+
+A run samples at most 50 rows from the dataset regardless of how many rows the dataset holds, so results reflect that sample rather than the full dataset.
+
+
+Once the run starts you can leave the tab and come back; it keeps going either way. See [Read optimization results](/docs/optimization/guides/read-optimization-results) for how to track it and read what it produces.
+
+## Stop a run
+
+While a run's status chip reads **Queue** or **Running**, its row carries a **Stop** control. Clicking it opens the **Stop optimization run** modal; confirm with **Stop Optimization** to cancel the run. Once a run finishes, fails, or is already stopped, the control is gone.
+
+## Restart a stopped run
+
+A stopped run's status chip in the grid reads **Cancelled**. Click its row in the Optimization tab to open its page. It shows the **Optimization Stopped** panel: 'The run was stopped before completion. Click below to start it again.', with a **Re-Run Optimization** button. Click it to open the **Re-run Optimization** drawer prefilled from the stopped run: the name gets a `- Rerun - ` suffix, and the column, optimizer, model, config, and evals are carried over. Review the fields and click **Start Optimization** to launch it as a new run from the beginning.
+
+## Dive deeper
+
+
+
+ Compare GEPA against the other optimizers and when to reach for each
+
+
+ Run the same kind of optimization from code with the agent-opt library
+
+
diff --git a/src/pages/docs/optimization/index.mdx b/src/pages/docs/optimization/index.mdx
index 0d2a896f..dddb2a62 100755
--- a/src/pages/docs/optimization/index.mdx
+++ b/src/pages/docs/optimization/index.mdx
@@ -1,42 +1,29 @@
---
-title: "Future AGI Optimization: Improve Prompts with Algorithms"
-description: "Iteratively improve prompts using evaluation-driven feedback and optimization algorithms for higher-quality, more consistent AI responses."
+title: "Overview"
+description: "Automatically rewrite and score prompt variations until one wins, using evals as the judge"
---
-## About
+## What is Optimization?
-**Optimization** is Future AGI's prompt improvement engine. It takes a prompt, runs it against your data, scores the outputs using evaluations, and iteratively generates better versions. Instead of manually tweaking prompts, you pick an algorithm and let it explore the prompt space systematically.
+**Optimization** points a run at a [prompt column](/docs/dataset/guides/run-a-prompt-on-every-row), scores rewrites against the [evals](/docs/evaluation) you pick to define what "good" means, and keeps the winner, using the optimizer algorithm you choose to generate and score those rewrites. Reach for it when a prompt is scoring badly and manual tweaking isn't converging.
-Future AGI supports 6 optimization algorithms: Random Search, Bayesian Search, Meta-Prompt, ProTeGi, GEPA, and PromptWizard. Each takes a different approach to exploring and improving prompts. You can run optimization from the platform UI or programmatically via the `agent-opt` Python SDK.
+Any of six optimizer algorithms can run against the same prompt column and eval set. They're alternative choices you swap on the same run, each searching for the new prompt differently, which is why [choosing an optimizer](/docs/optimization/concepts/choosing-an-optimizer) is worth doing deliberately.
-## How Optimization Connects to Other Features
+The same engine also runs from inside Simulation, where it optimizes an agent's prompt using results from a simulation test run instead of a dataset column. See [optimization in Simulation](/docs/simulation/concepts/optimization) for that path.
-- **Evaluation**: Optimization uses eval scores as its objective function. Better evals lead to better optimization. [Learn more](/docs/evaluation)
-- **Datasets**: Optimization runs against dataset rows. Your input/output pairs are the training ground. [Learn more](/docs/dataset)
-- **Experiments**: Compare optimized prompts against baselines using dataset experiments. [Learn more](/docs/dataset/features/experiments)
-
-## Getting Started
+## Keep exploring
-
- How optimization works, available algorithms, and how to choose the right one.
+
+ The model behind a run: dataset column, evals, and algorithm
+
+
+ How the six algorithms differ and which one fits your case
-
- Run optimization programmatically with the agent-opt library.
+
+ A run from the dataset's Optimization tab, step by step
-
- Run optimizations from the Future AGI UI with datasets and evals.
+
+ Parameters and behavior for each of the six algorithms
diff --git a/src/pages/docs/optimization/optimizers/bayesian-search.mdx b/src/pages/docs/optimization/optimizers/bayesian-search.mdx
deleted file mode 100644
index aed7211f..00000000
--- a/src/pages/docs/optimization/optimizers/bayesian-search.mdx
+++ /dev/null
@@ -1,138 +0,0 @@
----
-title: "Bayesian Search Optimizer for Few-Shot Prompt Tuning"
-description: "Use Bayesian optimization for few-shot prompt tuning: learns from trials to pick better example sets and configurations."
----
-
-Bayesian Search uses Bayesian optimization (via Optuna) to explore the space of few-shot prompt configurations. It learns from each trial to choose which examples and configurations to try next, so it converges faster than random search.
-
----
-
-## When to Use Bayesian Search
-
-
-
- - Few-shot learning tasks
- - Structured Q&A or classification
- - Limited evaluation budget
-
-
- - Tasks without examples in dataset
- - Purely zero-shot or very creative tasks
- - Tiny datasets (< 10 examples)
-
-
-
----
-
-## How It Works
-
-
-
- Set the range of few-shot examples (e.g. 2–8) and optional formatting.
-
-
- The optimizer suggests how many examples and which ones to use.
-
-
- Selected examples are formatted with the base prompt; outputs are scored on an eval subset.
-
-
- Results feed the next suggestion until the trial budget is used.
-
-
-
----
-
-## Basic Usage
-
-```python
-from fi.opt.optimizers import BayesianSearchOptimizer
-from fi.opt.datamappers import BasicDataMapper
-from fi.opt.base.evaluator import Evaluator
-
-# Setup evaluator
-evaluator = Evaluator(
- eval_template="summary_quality",
- eval_model_name="turing_flash",
- fi_api_key="your_key",
- fi_secret_key="your_secret"
-)
-
-# Setup data mapper
-data_mapper = BasicDataMapper(
- key_map={"input": "text", "output": "generated_output"}
-)
-
-# Create optimizer
-optimizer = BayesianSearchOptimizer(
- inference_model_name="gpt-4o-mini",
- n_trials=20,
- min_examples=2,
- max_examples=8
-)
-
-# Run optimization
-result = optimizer.optimize(
- evaluator=evaluator,
- data_mapper=data_mapper,
- dataset=dataset,
- initial_prompts=["Summarize: {text}"]
-)
-```
-
----
-
-## Parameters
-
-| Parameter | Type | Default | Description |
-|-----------|------|---------|-------------|
-| `min_examples` | int | 2 | Minimum few-shot examples per trial |
-| `max_examples` | int | 8 | Maximum few-shot examples per trial |
-| `allow_repeats` | bool | false | Same example can appear multiple times in few-shot block |
-| `fixed_example_indices` | List[int] | [] | Example indices always included (e.g. [0, 5]) |
-| `n_trials` | int | 10 | Number of configurations to try |
-| `seed` | int | 42 | Random seed |
-| `direction` | str | "maximize" | "maximize" for scores, "minimize" for loss |
-| `inference_model_name` | str | gpt-4o-mini | Model for generated outputs |
-| `example_template` | str | None | Template per example, e.g. "Q: {question}\nA: {answer}" |
-| `example_separator` | str | "\n" | String between examples in few-shot block |
-| `few_shot_position` | str | "append" | "append" or "prepend" |
-| `infer_example_template_via_teacher` | bool | false | Use teacher to infer example format (adds API cost) |
-| `teacher_model_name` | str | gpt-5 | Model for template inference when enabled |
-| `eval_subset_size` | int | None | Examples to evaluate per trial; None = full dataset |
-| `eval_subset_strategy` | str | "random" | "random", "first", or "all" |
-
----
-
-## Key concepts
-
-- **Result and history:** `result.final_score` and `result.best_generator.get_prompt_template()` give the best run. `result.history` holds per-trial scores and prompts for analysis.
-- **Template inference:** Set `infer_example_template_via_teacher=True` when you're unsure how to format examples; the teacher proposes a format. You can reuse that format in later runs with `example_template` to save cost.
-- **Fixed examples:** Use `fixed_example_indices=[0, 5]` to always include specific examples while the optimizer varies the rest.
-
-**Tips:** Start with `n_trials=10`, then 20–30 for production. Use `eval_subset_size=20` on large datasets. Template errors: ensure fields in `example_template` exist in your data. Plateaus: try `infer_example_template_via_teacher=True` or increase `max_examples`.
-
-**Research:** [A Bayesian approach for prompt optimization](https://arxiv.org/abs/2312.00471); used in DSPy and few-shot surveys.
-
----
-## **Underlying Research**
-
-Bayesian Search builds on established principles of Bayesian optimization, adapted for the unique challenges of prompt engineering.
-
-- **Core Concept**: The method is detailed in papers like "[A Bayesian approach for prompt optimization in pre-trained models](https://arxiv.org/abs/2312.00471)", which explores mapping discrete prompts to continuous embeddings for more efficient searching.
-- **Few-Shot Learning**: Its application in few-shot scenarios is highlighted by tools like Comet's OPik, which features a "Few-Shot Bayesian Optimizer".
-- **Advanced Implementations**: Recent research, such as "Searching for Optimal Solutions with LLMs via Bayesian Optimization (BOPRO)", investigates using Bayesian optimization to navigate complex LLM search spaces. The popular `BayesianOptimization` library on GitHub provides the foundational Gaussian process-based modeling.
-
-This approach is noted for its efficiency in prominent frameworks like DSPy and is recognized in surveys for its effectiveness in few-shot learning contexts.
-
----
-## Next steps
-
-
-
- For tasks that need deeper reasoning and full rewrites.
-
-
- See all optimization strategies.
-
-
diff --git a/src/pages/docs/optimization/optimizers/gepa.mdx b/src/pages/docs/optimization/optimizers/gepa.mdx
deleted file mode 100644
index 6f3cbb59..00000000
--- a/src/pages/docs/optimization/optimizers/gepa.mdx
+++ /dev/null
@@ -1,142 +0,0 @@
----
-title: "GEPA: Evolutionary Algorithm for Prompt Optimization"
-description: "GEPA (Genetic Pareto) is an evolutionary algorithm that evolves prompts over generations using reflection and mutation for complex optimization."
----
-
-GEPA (Genetic Pareto) is a powerful, state-of-the-art evolutionary algorithm that evolves a population of prompts over multiple generations. It uses a powerful "reflection" language model to analyze failures and provide feedback, which guides the mutation and evolution process toward creating better-performing prompts. It is designed for complex, high-stakes problems where achieving the best possible performance is critical.
-
----
-
-## When to Use GEPA
-
-
-
- - Complex, agentic AI systems
- - High-stakes optimization problems
- - Finding state-of-the-art prompts
- - Production-grade deployments
- - Effective alternative to Reinforcement Learning
-
-
-
- - Simple, straightforward tasks
- - Quick experiments or baseline testing
- - Projects with a low computational budget
- - Requires the external `gepa` library to be installed
-
-
-
----
-
-## How It Works
-
-GEPA uses a sophisticated evolutionary loop to systematically refine prompts. The process is managed by the external `gepa` library, which our optimizer adapts to.
-
-
-
- The process starts with a single `seed_candidate` prompt. An adapter is initialized to bridge our evaluation framework with the GEPA engine.
-
-
-
- GEPA's engine runs the current generation of prompts against the dataset. Our internal adapter calls our standard `Evaluator` to score the outputs, feeding the results back to GEPA.
-
-
-
- GEPA uses a powerful `reflection_lm` to analyze the evaluation results, especially the failures. It creates a "reflective dataset" that contains detailed feedback on why certain outputs were poor.
-
-
-
- The reflective dataset is used to guide the evolution process. The reflection model generates a new population of candidate prompts (mutations) that are specifically designed to avoid the failures of the previous generation.
-
-
-
- The new generation of prompts is evaluated, and the best-performing ones are selected to continue. This cycle repeats until a predefined budget (e.g., `max_metric_calls`) is exhausted, ensuring the process is efficient.
-
-
-
----
-
-## Basic Usage
-
-To use the GEPA optimizer, you need to provide two key models: one for reflection and one for generation.
-
-```python
-from fi.opt.optimizers import GEPAOptimizer
-from fi.opt.datamappers import BasicDataMapper
-from fi.opt.base.evaluator import Evaluator
-
-# 1. Setup the evaluator to score prompt performance
-evaluator = Evaluator(
- eval_template="summary_quality",
- eval_model_name="turing_flash",
- fi_api_key="your_key",
- fi_secret_key="your_secret"
-)
-
-# 2. Setup the data mapper
-data_mapper = BasicDataMapper(
- key_map={"input": "article", "output": "generated_output"}
-)
-
-# 3. Initialize the GEPA optimizer
-# The reflection_model should be a powerful LLM (e.g., GPT-4 Turbo)
-# The generator_model is the model your final prompt will use
-optimizer = GEPAOptimizer(
- reflection_model="gpt-4-turbo",
- generator_model="gpt-4o-mini"
-)
-
-# 4. Run the optimization
-# GEPA works towards a budget of total evaluations (max_metric_calls)
-result = optimizer.optimize(
- evaluator=evaluator,
- data_mapper=data_mapper,
- dataset=my_dataset,
- initial_prompts=["Summarize this article concisely: {article}"],
- max_metric_calls=200 # Total number of evaluations to perform
-)
-
-print(f"Best prompt found: {result.best_generator.get_prompt_template()}")
-print(f"Final score: {result.final_score:.4f}")
-```
-
----
-
-## Parameters
-
-| Parameter | Type | Default | Description |
-|-----------|------|---------|-------------|
-| `reflection_model` | str | required | Model for reflection and mutation (e.g. gpt-4-turbo, claude-3-opus) |
-| `generator_model` | str | gpt-4o-mini | Model for generated outputs (typically your production model) |
-| `max_metric_calls` | int | 150 | Total evaluation budget across all generations |
-
-**Key concepts:** `GEPAOptimizer` wraps the external `gepa` library; an internal adapter translates between our Evaluator and GEPA's engine (evaluate prompts, format reflection data). Install the `gepa` package to use this optimizer.
-
-**Tips:** Use a strong reflection model; set `max_metric_calls` (e.g. 100–150 for experiments). Library not found: install the external `gepa` library.
-
----
-
-## **Underlying Research**
-
-GEPA is based on recent advancements in evolutionary algorithms for prompt engineering, showing significant gains over traditional methods.
-
-- **Core Paper**: The method is detailed in "[GEPA: Reflective Prompt Evolution Can Outperform Reinforcement ...](https://arxiv.org/abs/2507.19457)", which demonstrates that it can outperform RL-based methods with far fewer evaluations.
-- **Efficiency**: As highlighted by the Databricks Blog, GEPA can lead to massive cost reductions for agent optimization. It is integrated into leading optimization frameworks like Opik and SuperOptiX.
-
----
-
-## Next steps
-
-
-
- For a different refinement approach
-
-
-
- See all optimization strategies.
-
-
\ No newline at end of file
diff --git a/src/pages/docs/optimization/optimizers/meta-prompt.mdx b/src/pages/docs/optimization/optimizers/meta-prompt.mdx
deleted file mode 100644
index 3ed7c428..00000000
--- a/src/pages/docs/optimization/optimizers/meta-prompt.mdx
+++ /dev/null
@@ -1,154 +0,0 @@
----
-title: "Meta-Prompt Optimizer: Teacher LLM Prompt Refinement"
-description: "The Meta-Prompt optimizer uses a teacher LLM for deep reasoning-based prompt refinement through systematic failure analysis and rewriting."
----
-
-Meta-Prompt uses a powerful teacher LLM to analyze how your prompt performs, understand why it fails on specific examples, formulate hypotheses about improvements, and completely rewrite the prompt. This approach is inspired by the `promptim` library and excels at tasks requiring deep reasoning.
-
----
-
-## When to Use Meta-Prompt
-
-
-
- - Complex reasoning tasks
- - Tasks where understanding failures helps
- - Refining well-scoped prompts
- - Deep iterative improvement
-
-
-
- - Quick experiments (slower)
- - Simple classification tasks
- - Very large datasets (costly)
- - Tasks with unclear failure patterns
-
-
-
----
-
-## How It Works
-
-Meta-Prompt follows a systematic analysis-and-rewrite cycle:
-
-
-
- Run the current prompt on a subset of your dataset and collect scores
-
-
-
- Focus on examples with low scores to understand what went wrong
-
-
-
- Teacher model analyzes failures and proposes a specific improvement theory
-
-
-
- Generate a complete new prompt implementing the hypothesis
-
-
-
- Continue for multiple rounds, building on previous insights
-
-
-
-**What the teacher sees (each round):** Current prompt; previous failed attempts (to avoid repeating mistakes); performance data (which examples failed and why); your task description.
-
-**What the teacher returns:** A hypothesis and an improved prompt, for example:
-
-```json
-{
- "hypothesis": "The prompt fails on complex multi-sentence texts because it doesn't specify a structure. Adding explicit instruction to identify main points first should improve clarity.",
- "improved_prompt": "First identify the 2-3 main points in the following text. Then write a single concise sentence that captures these points:\n\n{text}"
-}
-```
-
-
-Unlike optimizers that tweak parts of a prompt, Meta-Prompt rewrites the **entire** prompt each iteration based on deep analysis.
-
-
----
-
-## Basic Usage
-
-```python
-from fi.opt.optimizers import MetaPromptOptimizer
-from fi.opt.generators import LiteLLMGenerator
-from fi.opt.datamappers import BasicDataMapper
-from fi.opt.base.evaluator import Evaluator
-
-# Setup teacher model (use a powerful model for analysis)
-teacher = LiteLLMGenerator(
- model="gpt-4o",
- prompt_template="{prompt}"
-)
-
-# Setup evaluator
-evaluator = Evaluator(
- eval_template="summary_quality",
- eval_model_name="turing_flash",
- fi_api_key="your_key",
- fi_secret_key="your_secret"
-)
-
-# Setup data mapper
-data_mapper = BasicDataMapper(
- key_map={"input": "text", "output": "generated_output"}
-)
-
-# Create optimizer
-optimizer = MetaPromptOptimizer(
- teacher_generator=teacher
-)
-
-# Run optimization
-result = optimizer.optimize(
- evaluator=evaluator,
- data_mapper=data_mapper,
- dataset=dataset,
- initial_prompts=["Summarize this text: {text}"],
- task_description="Create concise, informative summaries",
- num_rounds=5,
- eval_subset_size=40
-)
-
-print(f"Improvement: {result.final_score:.2%}")
-print(f"Best prompt:\n{result.best_generator.get_prompt_template()}")
-```
-
----
-
-## Parameters
-
-| Parameter | Type | Default | Description |
-|-----------|------|---------|-------------|
-| `teacher_generator` | LiteLLMGenerator | required | Model for analysis and rewrites (e.g. gpt-4o, claude-3-opus) |
-| `task_description` | str | "I want to improve my prompt." | What the optimized prompt should achieve; more specific helps |
-| `num_rounds` | int | 5 | Analysis-and-rewrite iterations (passed to `optimize()`) |
-| `eval_subset_size` | int | 40 | Examples to evaluate each round (passed to `optimize()`) |
-
-**Key concepts:** Each round the teacher sees the current prompt, failed attempts, performance data, and your task description. It returns a hypothesis and an improved prompt. Unlike tweaking parts of a prompt, Meta-Prompt rewrites the **entire** prompt each iteration.
-
----
-
-## **Underlying Research**
-
-The Meta-Prompt optimizer is inspired by meta-learning and reflective AI systems, where a model improves its own processes.
-
-- **Meta-Learning**: The core idea is formalized in research like "[System Prompt Optimization with Meta-Learning](https://arxiv.org/abs/2505.09666)", which uses bilevel optimization. Another related work is "[metaTextGrad](https://arxiv.org/abs/2505.18524)", which optimizes both prompts and their surrounding structures.
-- **Industry Tools**: This reflective approach is used in tools like Google's Vertex AI Prompt Optimizer and is a key feature in advanced models for self-improvement.
-- **Frameworks**: The concept is explored in libraries like `promptim` and is classified in surveys as a leading LLM-driven optimization method.
-
----
-
-## Next steps
-
-
-
- For more systematic error analysis.
-
-
- See all optimization strategies.
-
-
\ No newline at end of file
diff --git a/src/pages/docs/optimization/optimizers/promptwizard.mdx b/src/pages/docs/optimization/optimizers/promptwizard.mdx
deleted file mode 100644
index aef236c8..00000000
--- a/src/pages/docs/optimization/optimizers/promptwizard.mdx
+++ /dev/null
@@ -1,147 +0,0 @@
----
-title: "PromptWizard: Multi-Stage Feedback-Driven Prompt Optimizer"
-description: "Learn about PromptWizard, a multi-stage feedback-driven optimizer that improves prompts through a cycle of mutation, critique, and refinement."
----
-
-PromptWizard is a feedback-driven optimizer that improves prompts through a multi-stage process. It first explores creative variations of a prompt using different "thinking styles," identifies the most promising candidates, critiques their failures, and then systematically refines them. It uses beam search to maintain and evolve the best-performing prompts over several iterations.
-
----
-
-## When to Use PromptWizard
-
-
-
- - Creative domains and content generation
- - Improving prompt style and meta-instructions
- - Complex tasks requiring reasoning
- - When you need a balance of exploration and refinement
-
-
-
- - Quick, simple optimizations
- - When teacher model quality is low
- - Projects with tight computational budgets
- - Tasks with very narrow, specific failure modes (ProTeGi may be better)
-
-
-
----
-
-## How It Works
-
-PromptWizard follows a sophisticated, multi-stage loop for a set number of `refine_iterations`. Each iteration aims to evolve the best prompt from the previous round.
-
-
-
- The optimizer takes the current best prompt and generates numerous creative variations. It uses a powerful teacher model and a list of diverse "thinking styles" (e.g., "Think step-by-step," "Analyze from different perspectives") to create a large pool of candidate prompts.
-
-
-
- All candidate prompts in the pool are evaluated against a subset of the dataset. Their performance is scored, and the top prompts are selected based on the `beam_size`. This ensures that only the most promising variations proceed.
-
-
-
- For each of the top-performing prompts, the optimizer identifies specific examples from the dataset where it performed poorly (i.e., received a low score). The teacher model then generates a detailed critique, explaining the likely reasons for failure.
-
-
-
- Using the original prompt, the failed examples, and the generated critique, the teacher model rewrites the prompt to address the identified weaknesses. This creates a new set of refined prompts.
-
-
-
- The refined prompts are scored again. The single best-performing prompt becomes the input for the next full iteration of the mutate-critique-refine cycle. This process repeats, progressively enhancing the prompt's quality.
-
-
-
----
-
-## Basic Usage
-
-```python
-from fi.opt.optimizers import PromptWizardOptimizer
-from fi.opt.generators import LiteLLMGenerator
-from fi.opt.datamappers import BasicDataMapper
-from fi.opt.base.evaluator import Evaluator
-
-# 1. Setup a powerful teacher model for the optimization process
-teacher = LiteLLMGenerator(
- model="gpt-4o",
- prompt_template="{prompt}"
-)
-
-# 2. Setup the evaluator to score prompt performance
-evaluator = Evaluator(
- eval_template="summary_quality",
- eval_model_name="turing_flash",
- fi_api_key="your_key",
- fi_secret_key="your_secret"
-)
-
-# 3. Setup the data mapper
-data_mapper = BasicDataMapper(
- key_map={"input": "article", "output": "generated_output"}
-)
-
-# 4. Initialize the PromptWizard optimizer
-optimizer = PromptWizardOptimizer(
- teacher_generator=teacher,
- mutate_rounds=3, # Number of mutation rounds per iteration
- refine_iterations=2, # Total number of refinement cycles
- beam_size=2 # Keep top 2 prompts for critique/refinement
-)
-
-# 5. Run the optimization
-result = optimizer.optimize(
- evaluator=evaluator,
- data_mapper=data_mapper,
- dataset=my_dataset,
- initial_prompts=["Summarize the following article: {article}"],
- task_description="Generate a concise, one-sentence summary of the article.",
- eval_subset_size=20
-)
-
-print(f"Best prompt found: {result.best_generator.get_prompt_template()}")
-print(f"Final score: {result.final_score:.4f}")
-```
-
----
-
-## Parameters
-
-| Parameter | Type | Default | Description |
-|-----------|------|---------|-------------|
-| `teacher_generator` | LiteLLMGenerator | required | Model for mutation, critique, and refinement (e.g. gpt-4o) |
-| `mutate_rounds` | int | 3 | Mutation calls per iteration; more = more diverse pool |
-| `refine_iterations` | int | 2 | Full cycles (Mutate → Score → Critique → Refine) |
-| `beam_size` | int | 1 | Top prompts to keep for critique and refinement |
-
-**Tips:** Use a strong teacher; start with `mutate_rounds=3`, `refine_iterations=2`. Slow: reduce those or `eval_subset_size`. Little improvement: make `task_description` more specific or try ProTeGi for clear failure patterns.
-
-**Vs ProTeGi:** PromptWizard explores first (mutate with "thinking styles") then refines; best for novel phrasings and style. ProTeGi is error-driven (fix specific failures); best when you have identifiable flaws to fix.
-
----
-
-## **Underlying Research**
-
-PromptWizard is based on the concept of self-evolving prompts, where an LLM iteratively improves its own instructions.
-
-- **Core Paper**: The framework is introduced in "[PromptWizard: Task-Aware Prompt Optimization Framework](https://arxiv.org/abs/2405.18369)" from Microsoft Research.
-- **Self-Evolution**: The underlying mechanism is detailed in "[Optimizing Prompts via Task-Aware, Feedback-Driven Self-Evolution](https://aclanthology.org/2025.findings-acl.1/)", which discusses the joint optimization of instructions and examples. The Microsoft Research Blog highlights this as a key direction for the future of prompt optimization.
-
----
-
-## Next steps
-
-
-
- For a more error-driven approach
-
-
-
- See all optimization strategies.
-
-
\ No newline at end of file
diff --git a/src/pages/docs/optimization/optimizers/protegi.mdx b/src/pages/docs/optimization/optimizers/protegi.mdx
deleted file mode 100644
index ea16a33a..00000000
--- a/src/pages/docs/optimization/optimizers/protegi.mdx
+++ /dev/null
@@ -1,151 +0,0 @@
----
-title: "ProTeGi: Prompt Optimization with Textual Gradients"
-description: "ProTeGi improves prompts by identifying failures, generating critiques, and applying targeted fixes using Textual Gradients for systematic optimization."
----
-
-ProTeGi (Prompt optimization with Textual Gradients) systematically improves prompts by identifying failure patterns, generating targeted critiques, and applying specific fixes. It uses beam search to maintain multiple candidate prompts and progressively refines them.
-
----
-
-## When to Use ProTeGi
-
-
-
- - Debugging specific failure modes
- - Systematic error correction
- - Tasks with clear failure patterns
- - Iterative refinement workflows
-
-
-
- - Quick experiments (multi-stage process)
- - Tasks where failures are random
- - Very small datasets
- - Budget-constrained projects
-
-
-
----
-
-## How It Works
-
-ProTeGi follows a structured expansion and selection process:
-
-
-
- Run current prompts and identify examples with low scores
-
-
-
- Teacher model analyzes failures and generates multiple specific critiques ("gradients")
-
-
-
- For each critique, generate improved prompt variations
-
-
-
- Evaluate all candidates and keep top N prompts
-
-
-
- Repeat expansion from the best performing prompts
-
-
-
-
-ProTeGi maintains a "beam" of candidate prompts throughout optimization, preventing premature convergence to local optima.
-
-
----
-
-## Basic Usage
-
-```python
-from fi.opt.optimizers import ProTeGi
-from fi.opt.generators import LiteLLMGenerator
-from fi.opt.datamappers import BasicDataMapper
-from fi.opt.base.evaluator import Evaluator
-
-# Setup teacher model
-teacher = LiteLLMGenerator(
- model="gpt-4o",
- prompt_template="{prompt}"
-)
-
-# Setup evaluator
-evaluator = Evaluator(
- eval_template="context_relevance",
- eval_model_name="turing_flash",
- fi_api_key="your_key",
- fi_secret_key="your_secret"
-)
-
-# Setup data mapper
-data_mapper = BasicDataMapper(
- key_map={"input": "question", "output": "generated_output"}
-)
-
-# Create optimizer
-optimizer = ProTeGi(
- teacher_generator=teacher,
- num_gradients=4,
- errors_per_gradient=4,
- prompts_per_gradient=1,
- beam_size=4
-)
-
-# Run optimization
-result = optimizer.optimize(
- evaluator=evaluator,
- data_mapper=data_mapper,
- dataset=dataset,
- initial_prompts=["Answer the question: {question}"],
- num_rounds=3,
- eval_subset_size=32
-)
-```
-
----
-
-## Parameters
-
-| Parameter | Type | Default | Description |
-|-----------|------|---------|-------------|
-| `teacher_generator` | LiteLLMGenerator | required | Model for critiques and improved prompts (e.g. gpt-4o) |
-| `num_gradients` | int | 4 | Critiques to generate per prompt |
-| `errors_per_gradient` | int | 4 | Failed examples shown to teacher per critique |
-| `prompts_per_gradient` | int | 1 | New prompts per critique (2–3 for more exploration) |
-| `beam_size` | int | 4 | Top prompts to keep each round |
-| `num_rounds` | int | 3 | Rounds (passed to `optimize()`) |
-| `eval_subset_size` | int | None | Examples per round; None = full dataset |
-
-**Tips:** Use a strong teacher; `beam_size` 3–4 is a good default. Plateau: increase `beam_size` or `num_gradients`. Slow: set `eval_subset_size=20` or reduce `beam_size`.
-
----
-
-## **Underlying Research**
-
-ProTeGi introduces a gradient-inspired approach to prompt optimization, adapting concepts from numerical optimization to natural language.
-
-- **Core paper:** [Automatic Prompt Optimization with "Gradient Descent" and Beam Search](https://arxiv.org/abs/2305.03495) details how to create "textual gradients" (critiques) to guide prompt improvement.
-- **Extensions:** [Momentum-Aided Gradient Descent Prompt Optimization](https://arxiv.org/abs/2410.19499) incorporates momentum to accelerate convergence.
-- **Classification:** In surveys on automatic prompt engineering, ProTeGi is categorized as a pioneering gradient-based method for error-driven refinement.
-
----
-
-## Next steps
-
-
-
- For exploration-first refinement
-
-
-
- See all optimization strategies.
-
-
\ No newline at end of file
diff --git a/src/pages/docs/optimization/optimizers/random-search.mdx b/src/pages/docs/optimization/optimizers/random-search.mdx
deleted file mode 100644
index 259615ba..00000000
--- a/src/pages/docs/optimization/optimizers/random-search.mdx
+++ /dev/null
@@ -1,134 +0,0 @@
----
-title: "Random Search Optimizer for Prompt Optimization Baseline"
-description: "Random Search is a simple, gradient-free method for establishing a baseline in prompt optimization by exploring random prompt variations."
----
-
-Random Search is a gradient-free method that generates a set of random variations of an initial prompt using a powerful "teacher" LLM. It then evaluates each variation against a dataset and selects the best-performing one. It's a fast, straightforward, and often surprisingly effective way to explore different prompt phrasings and establish a strong performance baseline.
-
----
-
-## When to Use Random Search
-
-
-
- - Establishing a quick baseline
- - Simple tasks like summarization or classification
- - Broad, unbiased exploration of the prompt space
- - Projects with a low computational budget
-
-
-
- - Complex, nuanced, or multi-step reasoning tasks
- - Directed, efficient optimization when failure modes are known
- - Tasks requiring highly structured or constrained prompts
- - Finding the absolute, state-of-the-art best prompt
-
-
-
----
-
-## How It Works
-
-The Random Search process is simple and effective, involving three main steps:
-
-
-
- You provide an initial prompt. The optimizer then uses a powerful `teacher_model` (like GPT-4o) to generate a specified `num_variations` of diverse rewrites of that prompt.
-
-
-
- The optimizer iterates through each generated variation. For each one, it generates outputs for all examples in your dataset and scores them using the provided evaluator.
-
-
-
- The variation that achieves the highest average score across the entire dataset is chosen as the best prompt. The process concludes, and this top-performing prompt is returned.
-
-
-
----
-
-## Basic Usage
-
-```python
-from fi.opt.optimizers import RandomSearchOptimizer
-from fi.opt.generators import LiteLLMGenerator
-from fi.opt.datamappers import BasicDataMapper
-from fi.opt.base.evaluator import Evaluator
-
-# 1. Define the generator with the initial prompt to be optimized
-initial_generator = LiteLLMGenerator(
- model="gpt-4o-mini",
- prompt_template="Summarize this article: {article}"
-)
-
-# 2. Setup the evaluator to score prompt performance
-evaluator = Evaluator(
- eval_template="summary_quality",
- eval_model_name="turing_flash",
- fi_api_key="your_key",
- fi_secret_key="your_secret"
-)
-
-# 3. Setup the data mapper
-data_mapper = BasicDataMapper(
- key_map={"input": "article", "output": "generated_output"}
-)
-
-# 4. Initialize the Random Search optimizer
-# It needs the generator to optimize, a powerful teacher model, and the number of variations to try.
-optimizer = RandomSearchOptimizer(
- generator=initial_generator,
- teacher_model="gpt-4o",
- num_variations=10
-)
-
-# 5. Run the optimization
-result = optimizer.optimize(
- evaluator=evaluator,
- data_mapper=data_mapper,
- dataset=my_dataset
-)
-
-print(f"Best prompt found: {result.best_generator.get_prompt_template()}")
-print(f"Final score: {result.final_score:.4f}")
-```
-
----
-
-## Parameters
-
-| Parameter | Type | Default | Description |
-|-----------|------|---------|-------------|
-| `generator` | BaseGenerator | required | Generator to optimize (prompt template is modified) |
-| `teacher_model` | str | gpt-5 | Model that generates variations (e.g. gpt-4o, claude-3-opus) |
-| `num_variations` | int | 5 | Number of prompt variations to generate and evaluate |
-| `teacher_model_kwargs` | dict | {} | Extra args for teacher (e.g. temperature for diversity) |
-
-**Tips:** Use a strong teacher; start with `num_variations=5` then 10–20. Similar scores: increase variations or check evaluator. Similar rewrites: raise temperature in `teacher_model_kwargs`.
-
----
-
-## **Underlying Research**
-
-Random search is a foundational technique in hyperparameter tuning, valued for its simplicity and surprising effectiveness.
-
-- **Baseline strength:** [Random Sampling as a Strong Baseline for Prompt Optimisation](https://arxiv.org/abs/2311.09569) shows that simple random sampling can be highly competitive for improving prompts.
-- **Use in toolkits:** It is often the first step in prompt optimization to explore the landscape and avoid local optima in the discrete, high-dimensional space of prompt engineering.
-
----
-
-## Next steps
-
-
-
- For more intelligent, learning-based exploration
-
-
-
- See all optimization strategies.
-
-
\ No newline at end of file
diff --git a/src/pages/docs/optimization/reference/optimizers/bayesian-search.mdx b/src/pages/docs/optimization/reference/optimizers/bayesian-search.mdx
new file mode 100644
index 00000000..a90b7c02
--- /dev/null
+++ b/src/pages/docs/optimization/reference/optimizers/bayesian-search.mdx
@@ -0,0 +1,121 @@
+---
+title: "Bayesian Search"
+description: "Parameters, defaults, and a runnable example for the Bayesian Search optimizer."
+---
+
+## What Bayesian Search tunes
+
+Bayesian Search tunes the few-shot examples in a prompt rather than its wording. It searches over how many examples to include and which ones, building a model of which configurations score well and spending its trial budget on the promising ones instead of trying every combination. That makes it a fit when the prompt's wording is already fine and the examples are what's left to tune. See [Choosing an optimizer](/docs/optimization/concepts/choosing-an-optimizer) for how it compares to Random Search and the rest.
+
+The candidate examples come from the same `dataset` you pass to `optimize()`: each trial borrows a handful of dataset rows and formats them as few-shot examples, so `min_examples` and `max_examples` bound how many rows a single trial can borrow. See [Optimize from the SDK](/docs/optimization/guides/optimize-from-the-sdk#build-the-dataset) for how to build one.
+
+## Parameters
+
+The **On-screen label** column is the field name shown when you run this optimizer from the UI; see [Run an optimization](/docs/optimization/guides/run-an-optimization) for the full form walkthrough. All three keys are required when you submit the optimization from the platform; in the SDK they're optional keyword arguments. The **Default** column gives the UI's prefilled value and the SDK constructor's fallback when the argument is omitted.
+
+| Parameter | On-screen label | Default | Description |
+|---|---|---|---|
+| `min_examples` | Min examples | 2 prefilled in the UI, 2 in the SDK | Minimum number of few-shot examples to include in a trial. Fewer examples means less context per trial and a cheaper run |
+| `max_examples` | Max examples | 4 prefilled in the UI, 8 in the SDK | Maximum number of few-shot examples to include in a trial. More examples means more context but a longer, costlier prompt per trial |
+| `n_trials` | No.of trials | 5 prefilled in the UI, 10 in the SDK | Number of configurations the optimizer tries. Raising it searches more configurations at the cost of more evaluator calls |
+
+
+The form rejects a submission where `min_examples` is greater than or equal to `max_examples`: the two have to be strictly ordered.
+
+
+This table covers only the parameters specific to Bayesian Search that appear in the UI. Two other groups of arguments show up in the code below but are documented in the [SDK reference](/docs/optimization/reference/sdk-api) instead:
+
+- `inference_model_name`, the constructor argument that sets which model generates completions during the search
+- the `optimize()` arguments shared by every optimizer: `evaluator`, `data_mapper`, `dataset`, and `early_stopping`
+
+`initial_prompts` isn't one of those shared arguments: it's required on this optimizer's `optimize()` call specifically.
+
+## Usage
+
+- Install: `pip install agent-opt`
+- Keys: get `fi_api_key` and `fi_secret_key` from [Admin Settings](/docs/admin-settings/api-keys) (or set `FI_API_KEY`/`FI_SECRET_KEY` as environment variables and drop them from the `Evaluator` call below)
+
+```python
+from fi.opt.optimizers import BayesianSearchOptimizer
+from fi.opt.datamappers import BasicDataMapper
+from fi.opt.base.evaluator import Evaluator
+
+# Dataset: each row can also be drawn as a few-shot example.
+# Keep max_examples at or below your row count.
+# Each row pairs the `article` input with a target `summary`, so the few-shot examples
+# the optimizer samples show the input to output pattern, not just inputs.
+my_dataset = [
+ {
+ "article": "The James Webb Space Telescope has captured its clearest images yet of a distant exoplanet's atmosphere, revealing traces of carbon dioxide and methane.",
+ "summary": "JWST found carbon dioxide and methane in a distant exoplanet's atmosphere.",
+ },
+ {
+ "article": "Researchers have discovered a new enzyme that breaks down PET plastic at room temperature, far faster than any previously known enzyme.",
+ "summary": "A newly discovered enzyme breaks down PET plastic at room temperature faster than any known before it.",
+ },
+ {
+ "article": "A team of engineers unveiled a compact fusion reactor prototype that sustained plasma for a record twelve minutes under laboratory conditions.",
+ "summary": "Engineers unveiled a compact fusion reactor that sustained plasma for a record twelve minutes.",
+ },
+ {
+ "article": "City officials broke ground on a new light rail line intended to cut downtown commute times by nearly half once completed in 2028.",
+ "summary": "City officials broke ground on a light rail line meant to cut downtown commute times nearly in half by 2028.",
+ },
+ {
+ "article": "A previously undocumented species of deep-sea octopus was filmed for the first time near hydrothermal vents off the coast of Costa Rica.",
+ "summary": "A previously undocumented deep-sea octopus species was filmed for the first time near hydrothermal vents off Costa Rica.",
+ },
+ # ... add more rows here
+]
+
+# Evaluator that scores each configuration.
+# "summary_quality" and "turing_flash" are just this example's choices; fi_api_key/fi_secret_key are placeholders (see Admin Settings above)
+evaluator = Evaluator(
+ eval_template="summary_quality",
+ eval_model_name="turing_flash",
+ fi_api_key="your_key",
+ fi_secret_key="your_secret"
+)
+
+# Maps generator output and dataset fields to what the evaluator expects
+# "article" here must match the dataset's key above and the {article} placeholder in initial_prompts below
+data_mapper = BasicDataMapper(
+ key_map={"input": "article", "output": "generated_output"}
+)
+
+# --- What's specific to Bayesian Search ---
+optimizer = BayesianSearchOptimizer(
+ min_examples=2,
+ max_examples=4,
+ n_trials=10, # this is the default; raise it to search more configurations
+ inference_model_name="gpt-4o-mini"
+)
+
+result = optimizer.optimize(
+ evaluator=evaluator,
+ data_mapper=data_mapper,
+ dataset=my_dataset,
+ initial_prompts=["Summarize this article: {article}"]
+)
+
+print(f"Final score: {result.final_score:.4f}")
+print(f"Best prompt:\n{result.best_generator.get_prompt_template()}")
+```
+
+`eval_template` accepts any of the [built-in evaluation templates](/docs/evaluation/builtin); `eval_model_name` accepts any of the [evaluator models](/docs/evaluation/concepts/evaluator-models).
+
+`result.best_generator` holds the original prompt wording plus the winning set of few-shot examples; `result.final_score` is that combination's average score from the evaluator.
+
+## Keep exploring
+
+
+
+ Install agent-opt, set your keys, and build the dataset this example needs
+
+
+ How optimizers, prompts, and runs fit together
+
+
+ The cheapest way to check how much headroom a prompt has
+
+
diff --git a/src/pages/docs/optimization/reference/optimizers/gepa.mdx b/src/pages/docs/optimization/reference/optimizers/gepa.mdx
new file mode 100644
index 00000000..9d017859
--- /dev/null
+++ b/src/pages/docs/optimization/reference/optimizers/gepa.mdx
@@ -0,0 +1,113 @@
+---
+title: "GEPA"
+description: "Parameters, defaults, and a runnable example for the GEPA optimizer."
+---
+
+## When to use GEPA
+
+Use this when you want broad exploration across many prompt variants under a fixed evaluation budget, rather than [Meta-Prompt](/docs/optimization/reference/optimizers/meta-prompt)'s one directed lineage of edits.
+
+GEPA evolves a population of candidate prompts across generations instead of revising one prompt in place. Each generation is evaluated against the dataset, and a separate reflection model reads the failures and writes the next generation of candidates based on what went wrong.
+
+GEPA takes two models: `reflection_model` writes the new prompts, and `generator_model` is the model GEPA runs each candidate prompt on during optimization to produce the output that gets scored, and it's also the model the optimized prompt is meant to run on afterward; it defaults to `gpt-4o-mini`.
+
+## Parameters
+
+`evaluator`, `data_mapper`, and `dataset`, also passed in the example below, are shared by every optimizer's `optimize()` call and covered in the [SDK reference](/docs/optimization/reference/sdk-api).
+
+| Parameter | Set in | On-screen label | Default | Description |
+|---|---|---|---|---|
+| `reflection_model` | `GEPAOptimizer()` | - | required | Model that analyses failures and writes the next generation of candidate prompts |
+| `generator_model` | `GEPAOptimizer()` | - | gpt-4o-mini | Model GEPA runs each candidate on during optimization |
+| `initial_prompts` | `optimize()` | - | required | List of starting prompts (see note below: only the first is used) |
+| `max_metric_calls` | `optimize()` | Max Metric Calls | 40 prefilled in the UI, 150 in the SDK | Total evaluation budget across all generations |
+
+GEPA seeds from the first prompt in `initial_prompts` and ignores the rest, so passing more than one silently discards the extras.
+
+GEPA runs to a fixed budget of evaluations rather than a fixed number of rounds; `max_metric_calls` caps the total number of evaluations across the whole run:
+
+- How many generations `max_metric_calls` buys shrinks as your dataset grows
+- Raise `max_metric_calls` above the SDK's default of 150, or the platform's prefilled 40, to let GEPA work through more generations before stopping
+- Lower it for a cheaper, shallower run. Each unit is one scored row: a generator call plus the evaluator call you pay for, so cost and runtime scale roughly linearly with the value you set
+
+## Usage
+
+```bash
+pip install agent-opt
+```
+
+Then get `fi_api_key` and `fi_secret_key` from [Admin Settings](/docs/admin-settings/api-keys) (or set `FI_API_KEY`/`FI_SECRET_KEY` as environment variables and drop them from the `Evaluator` call below). This example builds an `Evaluator` and `BasicDataMapper` the same way every optimizer does; see the SDK reference above for their full constructors.
+
+Two rows are enough to sanity-check the code path.
+
+```python
+from fi.opt.optimizers import GEPAOptimizer
+from fi.opt.datamappers import BasicDataMapper
+from fi.opt.base.evaluator import Evaluator
+
+# Dataset: a plain list of dicts, one per example the optimizer scores the prompt against
+dataset = [
+ {
+ "article": "The James Webb Space Telescope has captured its clearest images yet of a distant exoplanet's atmosphere, revealing traces of carbon dioxide and methane.",
+ },
+ {
+ "article": "Researchers have discovered a new enzyme that breaks down PET plastic at room temperature, far faster than any previously known enzyme.",
+ },
+ # ... more rows
+]
+
+# Evaluator that scores each candidate prompt.
+# eval_template options: /docs/evaluation/concepts/eval-templates
+# eval_model_name options: /docs/evaluation/concepts/evaluator-models
+evaluator = Evaluator(
+ eval_template="summary_quality",
+ eval_model_name="turing_flash",
+ fi_api_key="your_key",
+ fi_secret_key="your_secret"
+)
+
+# Maps generator output and dataset fields to what the evaluator expects.
+# "generated_output" isn't a dataset key you provide: GEPA writes each candidate's
+# output there after running it through generator_model, for the evaluator to score.
+data_mapper = BasicDataMapper(
+ key_map={"input": "article", "output": "generated_output"}
+)
+
+optimizer = GEPAOptimizer(
+ reflection_model="gpt-4-turbo",
+ generator_model="gpt-4o-mini"
+)
+
+result = optimizer.optimize(
+ evaluator=evaluator,
+ data_mapper=data_mapper,
+ dataset=dataset,
+ # GEPA seeds from the first prompt only; any others in this list are ignored.
+ initial_prompts=["Summarize this article concisely: {article}"],
+ max_metric_calls=150
+)
+
+print(f"Final score: {result.final_score:.4f}")
+print(f"Best prompt:\n{result.best_generator.get_prompt_template()}")
+
+# Example output:
+# Final score: 0.8700
+# Best prompt:
+# Summarize this article in one sentence, focusing on the key finding: {article}
+```
+
+A successful run prints the final score followed by the best prompt, as in the two `print` calls above. `result` carries other fields beyond `final_score` and `best_generator`; see the SDK reference above for the full list. See [Optimize from the SDK](/docs/optimization/guides/optimize-from-the-sdk) for how to take the winning prompt into production.
+
+## Keep exploring
+
+
+
+ Install agent-opt, set your keys, and build the dataset this example needs
+
+
+ How optimizers, prompts, and runs fit together
+
+
+ A learning-based optimizer for few-shot prompt tuning
+
+
diff --git a/src/pages/docs/optimization/reference/optimizers/index.mdx b/src/pages/docs/optimization/reference/optimizers/index.mdx
new file mode 100644
index 00000000..f4dc06f7
--- /dev/null
+++ b/src/pages/docs/optimization/reference/optimizers/index.mdx
@@ -0,0 +1,46 @@
+---
+title: "Optimizers"
+description: "The parameter fields, defaults, and agent-opt class for every optimizer, in one table."
+---
+
+Six optimizers are available, each suited to a different problem shape. See [Choosing an optimizer](/docs/optimization/concepts/choosing-an-optimizer) for how to pick one, or [Understanding optimization](/docs/optimization/concepts/understanding-optimization) for how the optimization loop works.
+
+## Parameters by optimizer
+
+The parameter set below describes what you pass in the platform's optimization request when you create a run, not arguments to the class constructor; see [Run an optimization](/docs/optimization/guides/run-an-optimization) for one optimizer's parameters going into a request end to end. Optimization Objective (`task_description`) is present on every optimizer's form, though nothing is prefilled into it.
+
+| Optimizer | Class | Use it when | Parameters (label / code key / default) |
+|---|---|---|---|
+| [Random Search](/docs/optimization/reference/optimizers/random-search) | `RandomSearchOptimizer` | First look at a prompt's headroom | Number Variations (`num_variations`), 3 |
+| [Bayesian Search](/docs/optimization/reference/optimizers/bayesian-search) | `BayesianSearchOptimizer` | Tuning few-shot examples on solid instructions | Min examples (`min_examples`), 2 Max examples (`max_examples`), 4 No.of trials (`n_trials`), 5 |
+| [ProTeGi](/docs/optimization/reference/optimizers/protegi) | `ProTeGi` | Parallel candidate fixes from failing examples | Beam size (`beam_size`), 4 Number of gradients (`num_gradients`), 4 Errors per gradient (`errors_per_gradient`), 4 Prompts per gradient (`prompts_per_gradient`), 1 Number of Rounds (`num_rounds`), 3 |
+| [Meta-Prompt](/docs/optimization/reference/optimizers/meta-prompt) | `MetaPromptOptimizer` | One prompt refined over rounds, not parallel candidates | Optimization Objective (`task_description`) Number of Rounds (`num_rounds`), 4 |
+| [PromptWizard](/docs/optimization/reference/optimizers/promptwizard) | `PromptWizardOptimizer` | Wording itself, not one instruction, is the ceiling | Mutated Rounds (`mutate_rounds`), 3 Refined Iterations (`refine_iterations`), 2 Beam size (`beam_size`), 2 |
+| [GEPA](/docs/optimization/reference/optimizers/gepa) | `GEPAOptimizer` | Widest search of the six, 40 metric calls prefilled | Max Metric Calls (`max_metric_calls`), 40 |
+
+Every parameter listed above must be passed on the optimization request. The values shown are what the form prefills when you pick that optimizer, and you can change any of them before you start the run. A request missing any listed parameter, or carrying a parameter that isn't listed for that optimizer and isn't `task_description`, is rejected before the run starts; see [Common errors and fixes](/docs/optimization/troubleshooting#common-errors-and-fixes) for the exact error.
+
+`task_description` is the one parameter allowed outside a row's own list: it can additionally be passed on any optimizer's request, even where its row above doesn't list it for that optimizer, and Meta-Prompt is the one optimizer that requires it as part of its own parameter set.
+
+## Keep exploring
+
+
+
+ Unguided variations on the wording
+
+
+ Tunes which few-shot examples are used, not the wording
+
+
+ Several fixes drawn from failures, explored in parallel
+
+
+ One prompt rewritten from failures, iterated over rounds
+
+
+ Mutates the wording, then critiques and refines the result
+
+
+ Evolutionary search with a budget set in metric calls, 40 prefilled
+
+
diff --git a/src/pages/docs/optimization/reference/optimizers/meta-prompt.mdx b/src/pages/docs/optimization/reference/optimizers/meta-prompt.mdx
new file mode 100644
index 00000000..4d73c270
--- /dev/null
+++ b/src/pages/docs/optimization/reference/optimizers/meta-prompt.mdx
@@ -0,0 +1,107 @@
+---
+title: "Meta-Prompt"
+description: "Parameters, defaults, and a runnable example for the Meta-Prompt optimizer."
+---
+
+## When to use Meta-Prompt
+
+Meta-Prompt has a teacher model analyze each round's failures and rewrite the whole prompt, rather than patching individual parts of it. Use it when a prompt needs rethinking rather than incremental tuning.
+
+## Parameters
+
+The On-screen label column maps the SDK parameter to the platform UI's field label; parameters without one aren't exposed there. The Required column reflects the SDK call signature only, not the platform form's own required fields. The Default column is scoped the same way: in the UI, Number of Rounds is required and not prefilled, since the form starts with an empty configuration, so `5` is the SDK/backend fallback that only applies when `num_rounds` is left out of `optimize()`.
+
+| Parameter | Set in | Required (SDK) | On-screen label | Default | Description |
+|---|---|---|---|---|---|
+| `teacher_generator` | `MetaPromptOptimizer()` | Yes | - | - | The `LiteLLMGenerator` that analyzes each round's failures and rewrites the prompt |
+| `task_description` | `optimize()` | No | Optimization Objective | `"I want to improve my prompt."` | What the optimized prompt should achieve |
+| `num_rounds` | `optimize()` | No | Number of Rounds | Required in the UI; 5 in the SDK | Number of analysis-and-rewrite iterations the teacher model runs |
+| `eval_subset_size` | `optimize()` | No | - | 40 | Number of dataset rows sampled for evaluation each round (capped to the dataset size) |
+| `initial_prompts` | `optimize()` | Yes | - | - | The first prompt in `initial_prompts` to optimize |
+
+The teacher receives the meta-prompt, built from the current prompt, the task description, and the round's failures, and rewrites the prompt in response. In round 1, the current prompt is the first prompt in `initial_prompts`; from round 2 on, it's the teacher's own last rewrite, alongside the earlier attempts that already scored worse.
+
+The teacher is a `LiteLLMGenerator`. Weigh a stronger model against a cheaper one the same way you would for the evaluator: better rewrites versus lower per-round cost.
+
+The example below uses `gpt-4o-mini`.
+
+`task_description` is not specific to Meta-Prompt: every optimizer accepts it alongside its own parameters; for Meta-Prompt, it's the goal statement the teacher rewrites the prompt against. The SDK default above only applies if you omit the argument to `optimize()`. A run started from the platform sends a request that must carry both `task_description` and `num_rounds` keys.
+
+`optimize()` also takes `evaluator`, `data_mapper`, and `dataset`, shared by every optimizer's `optimize()` call and covered in the [SDK reference](/docs/optimization/reference/sdk-api).
+
+Raising `num_rounds` gives the teacher model more analyze-and-rewrite cycles before settling, at the cost of one teacher-model call per extra round, plus one generator call and one evaluator call for each row in that round's eval subset (`min(len(dataset), eval_subset_size)` rows, so up to 40 by default). Start at the default of 5 and raise it if the score is still improving by the last round; lower it for a quick check.
+
+## Usage
+
+Meta-Prompt is available from the [platform UI](/docs/optimization/guides/run-an-optimization) as well as the Python SDK below.
+
+```bash
+pip install agent-opt
+```
+
+This installs the `fi.opt` namespace used in the imports below. Get `fi_api_key` and `fi_secret_key` from [Admin Settings](/docs/admin-settings/api-keys) (or set `FI_API_KEY`/`FI_SECRET_KEY` as environment variables and drop them from the `Evaluator` call below).
+
+```python
+from fi.opt.optimizers import MetaPromptOptimizer
+from fi.opt.generators import LiteLLMGenerator
+from fi.opt.datamappers import BasicDataMapper
+from fi.opt.base.evaluator import Evaluator
+
+# Dataset: a list of dicts, one per example. Keys must cover whatever the
+# prompt template and key_map below reference, here just "article".
+# See "Build the dataset" in /docs/optimization/guides/optimize-from-the-sdk.
+my_dataset = [
+ {"article": "The James Webb Space Telescope has captured its clearest images yet of a distant exoplanet's atmosphere, revealing traces of carbon dioxide and methane."},
+ {"article": "Researchers have discovered a new enzyme that breaks down PET plastic at room temperature, far faster than any previously known enzyme."},
+]
+
+# Teacher model that analyzes failures and rewrites the prompt. Its
+# prompt_template must be the passthrough "{prompt}": the optimizer sends
+# the whole meta-prompt through the "prompt" key, not the dataset's own keys.
+teacher_generator = LiteLLMGenerator(
+ model="gpt-4o-mini",
+ prompt_template="{prompt}"
+)
+
+# Evaluator that scores each rewrite
+# eval_template and eval_model_name options: /docs/evaluation/builtin and /docs/evaluation/concepts/evaluator-models
+evaluator = Evaluator(
+ eval_template="summary_quality",
+ eval_model_name="turing_flash",
+ fi_api_key="your_key",
+ fi_secret_key="your_secret"
+)
+
+# Maps generator output and dataset fields to what the evaluator expects
+data_mapper = BasicDataMapper(
+ key_map={"input": "article", "output": "generated_output"}
+)
+
+optimizer = MetaPromptOptimizer(teacher_generator=teacher_generator)
+
+result = optimizer.optimize(
+ initial_prompts=["Summarize this article: {article}"],
+ task_description="Create concise, informative summaries",
+ num_rounds=5,
+ evaluator=evaluator,
+ data_mapper=data_mapper,
+ dataset=my_dataset
+)
+
+# Read the optimized prompt and its score off the result
+print(f"Final score: {result.final_score:.4f}")
+print(f"Best prompt:\n{result.best_generator.get_prompt_template()}")
+```
+
+A successful run prints the final score followed by the rewritten prompt, as in the two `print` calls above. `result` carries other fields beyond `final_score` and `best_generator`; see the [SDK reference](/docs/optimization/reference/sdk-api) for the full list. If it errors instead, check that `key_map` in `data_mapper` matches both your dataset's field names and the evaluator's expected keys, that `eval_template` is a valid template name, and that your API credentials are correct.
+
+## Keep exploring
+
+
+
+ How optimizers, prompts, and runs fit together
+
+
+ The cheapest way to check how much headroom a prompt has
+
+
diff --git a/src/pages/docs/optimization/reference/optimizers/promptwizard.mdx b/src/pages/docs/optimization/reference/optimizers/promptwizard.mdx
new file mode 100644
index 00000000..2bc1699d
--- /dev/null
+++ b/src/pages/docs/optimization/reference/optimizers/promptwizard.mdx
@@ -0,0 +1,129 @@
+---
+title: "PromptWizard"
+description: "Parameters, defaults, and a runnable example for the PromptWizard optimizer."
+---
+
+## When to use PromptWizard
+
+PromptWizard suits open-ended tasks where the right framing for a prompt is not obvious upfront. It works by mutating a prompt into several different framings, then critiquing and refining the best of those framings over a set number of iterations, rather than reacting to specific failures the way [ProTeGi](/docs/optimization/reference/optimizers/protegi) does.
+
+PromptWizard runs from the SDK below, or from the platform's Run Optimization drawer; see [Run an optimization](/docs/optimization/guides/run-an-optimization) for the UI walkthrough.
+
+## Parameters
+
+| Parameter | On-screen label | Default | Description |
+|---|---|---|---|
+| `teacher_generator` | - | required, no default | Generator used for critique and refinement. Its `prompt_template` must be the passthrough `"{prompt}"`; PromptWizard fills it with its own critique-and-refine prompts, not your task prompt. Candidate prompts are run against your dataset by a generator PromptWizard manages internally, not by `teacher_generator` and not by a second generator you supply |
+| `mutate_rounds` | Mutated Rounds | 3 | Number of mutation rounds used to generate prompt variations |
+| `refine_iterations` | Refined Iterations | 2 | Number of full mutate, score, and refine cycles run on the best candidates; raising it repeats the mutation rounds again each cycle, not just the refine step |
+| `beam_size` | Beam size | 2 prefilled in the UI, 1 in the SDK | Number of top-scoring prompts carried forward at each round |
+
+This table covers only PromptWizard's own tuning knobs. `evaluator`, `data_mapper`, `dataset`, and `initial_prompts` (required on every `optimize()` call) are shared by every optimizer's `optimize()` call and are covered in the [SDK reference](/docs/sdk/optimization).
+
+Raising `mutate_rounds`, `refine_iterations`, or `beam_size` makes PromptWizard explore or refine more before it settles, at the cost of more generator and evaluator calls per run.
+
+
+Keep `mutate_rounds`, `refine_iterations`, and `beam_size` low for a quick pass. Raise them for a second pass, for example `mutate_rounds=5, refine_iterations=3, beam_size=2`.
+
+
+
+If you're porting a `beam_size` value from [ProTeGi](/docs/optimization/reference/optimizers/protegi), note that in the SDK constructor its default is 4, versus 1 for PromptWizard's constructor. This is a constructor-only comparison: PromptWizard's own on-screen default for this field is 2, not 1.
+
+
+## Usage
+
+Before running the example below:
+
+- **Install**: `pip install agent-opt`
+- **FI keys**: get `fi_api_key` and `fi_secret_key` from [Admin Settings](/docs/admin-settings/api-keys), or set `FI_API_KEY`/`FI_SECRET_KEY` as environment variables and drop them from the `Evaluator` call below
+- **Model key**: export `OPENAI_API_KEY` as an environment variable, since the example below passes `gpt-4o-mini` to `LiteLLMGenerator`; there's no field in the code to put it in
+
+The dataset below is a small inline list of dicts; see [Optimize from the SDK](/docs/optimization/guides/optimize-from-the-sdk#build-the-dataset) for loading your own data instead.
+
+```python
+from fi.opt.optimizers import PromptWizardOptimizer
+from fi.opt.generators import LiteLLMGenerator
+from fi.opt.datamappers import BasicDataMapper
+from fi.opt.base.evaluator import Evaluator
+
+# Dataset: a list of dicts, one per example. Keys must cover whatever the
+# prompt template and key_map below reference, here just "article".
+my_dataset = [
+ {"article": "The James Webb Space Telescope has captured its clearest images yet of a distant exoplanet's atmosphere, revealing traces of carbon dioxide and methane."},
+ {"article": "Researchers have discovered a new enzyme that breaks down PET plastic at room temperature, far faster than any previously known enzyme."},
+]
+
+# Teacher model used for critique and refinement; see teacher_generator in
+# the table above. Your task prompt is passed to initial_prompts on
+# optimize() below.
+generator = LiteLLMGenerator(
+ model="gpt-4o-mini",
+ prompt_template="{prompt}"
+)
+
+# Evaluator that scores each candidate prompt.
+# eval_template: see the built-in templates linked below
+# eval_model_name: see the evaluator models linked below
+evaluator = Evaluator(
+ eval_template="summary_quality",
+ eval_model_name="turing_flash",
+ fi_api_key="your_key",
+ fi_secret_key="your_secret"
+)
+
+# Maps generator output and dataset fields to what the evaluator expects.
+# "article" must match a key in my_dataset; "generated_output" is filled
+# in automatically by the optimizer, not a dataset field.
+data_mapper = BasicDataMapper(
+ key_map={"input": "article", "output": "generated_output"}
+)
+
+# mutate_rounds, refine_iterations, and beam_size here match the defaults
+# in the table above and can be omitted; shown so they're easy to change.
+optimizer = PromptWizardOptimizer(
+ teacher_generator=generator,
+ mutate_rounds=3,
+ refine_iterations=2,
+ beam_size=1
+)
+
+result = optimizer.optimize(
+ evaluator=evaluator,
+ data_mapper=data_mapper,
+ dataset=my_dataset,
+ # initial_prompts holds the starting prompt PromptWizard mutates and refines
+ initial_prompts=["Summarize this article: {article}"]
+)
+
+# Read the optimized prompt and its score off the result
+print(f"Final score: {result.final_score:.4f}")
+print(f"Best prompt:\n{result.best_generator.get_prompt_template()}")
+```
+
+`eval_template` accepts any of the [built-in evaluation templates](/docs/evaluation/builtin); `eval_model_name` accepts any of the [evaluator models](/docs/evaluation/concepts/evaluator-models).
+
+A successful run prints something like:
+
+```
+Final score: 0.8214
+Best prompt:
+Summarize this article in 2-3 sentences, covering the main finding and its significance.
+```
+
+The exact score and wording vary by run. If a `key_map` value doesn't match a field in the dataset (here, `article`), nothing raises or names the mismatch: the field is silently missing from what the evaluator sees. Make sure every value in `key_map` matches a key present in your dataset's dicts.
+
+`result` holds more than `final_score` and `best_generator`; see [OptimizationResult](/docs/optimization/reference/sdk-api#optimizationresult) for the full field list, and [Read the result](/docs/optimization/guides/optimize-from-the-sdk#read-the-result) for how to take `best_generator`'s prompt into production.
+
+## Keep exploring
+
+
+
+ Install agent-opt, set your keys, and build the dataset this example needs
+
+
+ How optimizers, prompts, and runs fit together
+
+
+ A learning-based optimizer for few-shot prompt tuning
+
+
diff --git a/src/pages/docs/optimization/reference/optimizers/protegi.mdx b/src/pages/docs/optimization/reference/optimizers/protegi.mdx
new file mode 100644
index 00000000..80972882
--- /dev/null
+++ b/src/pages/docs/optimization/reference/optimizers/protegi.mdx
@@ -0,0 +1,136 @@
+---
+title: "ProTeGi"
+description: "Parameters, defaults, and a runnable example for the ProTeGi optimizer."
+---
+
+## When to use ProTeGi
+
+Use this when a prompt is mostly right but has known, specific failure patterns, and you want several candidate fixes explored in parallel rather than one rewrite committed to at a time.
+
+ProTeGi reads the rows that scored badly, turns them into textual criticism, and applies targeted edits to the prompt. Each piece of criticism is called a **gradient**. The model that writes the gradients and the revised prompts is the **teacher model**, passed to the `teacher_generator` argument on the constructor, separately from the tuning parameters in the table below. ProTeGi keeps several revised candidates alive at once across rounds rather than committing to a single rewrite.
+
+## Parameters
+
+| Parameter | Set in | On-screen label | Default | Description |
+|---|---|---|---|---|
+| `beam_size` | `ProTeGi()` | Beam size | 4 prefilled in the UI, 4 in the SDK | Number of top-scoring candidate prompts kept alive each round |
+| `num_gradients` | `ProTeGi()` | Number of gradients | 4 prefilled in the UI, 4 in the SDK | Number of textual critiques generated from the failed rows |
+| `errors_per_gradient` | `ProTeGi()` | Errors per gradient | 4 prefilled in the UI, 4 in the SDK | Number of failed rows shown to the teacher model per critique |
+| `prompts_per_gradient` | `ProTeGi()` | Prompts per gradient | 1 prefilled in the UI, 1 in the SDK | Number of revised prompts generated per critique |
+| `num_rounds` | `optimize()` | Number of Rounds | 3 prefilled in the UI, 3 in the SDK | Number of rounds of critique and revision |
+| `teacher_generator` | `ProTeGi()` | - | Required | Teacher model that writes the gradients and revised prompts (not a tuning parameter) |
+| `initial_prompts` | `optimize()` | - | Required | Starting prompt(s) ProTeGi refines (not a tuning parameter) |
+
+The On-screen label column maps each SDK parameter to its field in the [platform UI](/docs/optimization/guides/run-an-optimization), where all five tuning parameters are required and prefilled with the values above. The Default column's SDK values apply only when you build a `ProTeGi()` call yourself and leave the argument out. `evaluator`, `data_mapper`, and `dataset`, also passed to `optimize()` in the example below, are shared by every optimizer and covered in the [SDK reference](/docs/optimization/reference/sdk-api), so they're left out of this table.
+
+Within a round, `num_gradients`, `errors_per_gradient`, and `prompts_per_gradient` multiply: each gradient draws on `errors_per_gradient` failed rows and produces `prompts_per_gradient` revised prompts, and the round's candidate count scales with `beam_size x num_gradients x prompts_per_gradient`. Raising any of them scales up that round's work, and `num_rounds` repeats it again each round.
+
+- If a run is too slow, lower `prompts_per_gradient` or `errors_per_gradient` first
+- If a run is too shallow, raise `num_gradients` or `num_rounds`
+- `beam_size` doesn't add work in round 1, since that round only expands the starting prompt(s); from round 2 on, expansion loops over the whole beam, so raising `beam_size` multiplies every later round's work by the same amount
+
+## Usage
+
+Before running this example:
+
+- Install the SDK: `pip install agent-opt`
+- Get an `FI_API_KEY` and `FI_SECRET_KEY` pair (see [API keys](/docs/admin-settings/api-keys) for where to get them). Pass them as environment variables or, as below, directly into `Evaluator`
+- Need a dataset to score against? See [Optimize from the SDK](/docs/optimization/guides/optimize-from-the-sdk) for how to build one like the one used below
+
+This example builds a starting prompt, tunes it against a small dataset over `num_rounds` rounds of critique and revision, and reads back the winning prompt and its score.
+
+```python
+from fi.opt.optimizers import ProTeGi # the class is ProTeGi, not ProTeGiOptimizer
+from fi.opt.generators import LiteLLMGenerator
+from fi.opt.datamappers import BasicDataMapper
+from fi.opt.base.evaluator import Evaluator
+
+# Dataset: a plain list of dicts, one per example the optimizer scores the prompt against
+dataset = [
+ {
+ "article": "The James Webb Space Telescope has captured its clearest images yet of a distant exoplanet's atmosphere, revealing traces of carbon dioxide and methane.",
+ "target_summary": "JWST detected carbon dioxide and methane in a distant exoplanet's atmosphere.",
+ },
+ {
+ "article": "Researchers have discovered a new enzyme that breaks down PET plastic at room temperature, far faster than any previously known enzyme.",
+ "target_summary": "A newly discovered enzyme breaks down PET plastic much faster than before.",
+ },
+ # ... more rows
+]
+
+# Teacher model that writes the gradients and revised prompts.
+# Its prompt_template is filled with ProTeGi's own critique and
+# revision instructions at runtime, so it should just pass them
+# through: set it to "{prompt}" regardless of your task. Your
+# starting prompt goes in initial_prompts on optimize() below,
+# not here.
+teacher_generator = LiteLLMGenerator(
+ model="gpt-4o-mini",
+ prompt_template="{prompt}"
+)
+
+# Evaluator that scores each revised candidate
+evaluator = Evaluator(
+ eval_template="summary_quality",
+ eval_model_name="turing_flash",
+ fi_api_key="your_key",
+ fi_secret_key="your_secret"
+)
+
+# Maps generator output and dataset fields to what the evaluator expects.
+# "generated_output" is the generator's fixed output key; "article" is
+# the dataset field from this example and should match your own data.
+data_mapper = BasicDataMapper(
+ key_map={"input": "article", "output": "generated_output"}
+)
+
+# beam_size, num_gradients, errors_per_gradient, and prompts_per_gradient
+# here match the defaults in the table above and can be omitted; shown so
+# they're easy to change.
+optimizer = ProTeGi(
+ teacher_generator=teacher_generator,
+ beam_size=4,
+ num_gradients=4,
+ errors_per_gradient=4,
+ prompts_per_gradient=1
+)
+
+# initial_prompts holds the starting prompt(s) ProTeGi refines.
+# num_rounds also matches the default in the table above and can be omitted.
+result = optimizer.optimize(
+ evaluator=evaluator,
+ data_mapper=data_mapper,
+ dataset=dataset,
+ initial_prompts=["Summarize this article: {article}"],
+ num_rounds=3
+)
+
+print(f"Final score: {result.final_score:.4f}")
+print(f"Best prompt:\n{result.best_generator.get_prompt_template()}")
+```
+
+`eval_template` and `eval_model_name` are real built-in names; see [eval templates](/docs/evaluation/concepts/eval-templates) and [evaluator models](/docs/evaluation/concepts/evaluator-models) for the full lists.
+
+A successful run prints something like:
+
+```
+Final score: 0.8532
+Best prompt:
+Summarize this article in 2-3 sentences, covering the main finding and its significance: {article}
+```
+
+The exact score and wording vary by run. If it errors instead, check that `key_map` in `data_mapper` matches your dataset's field names, that `eval_template` is a valid template name, and that the keys line up with what the evaluator expects.
+
+## Keep exploring
+
+
+
+ Install agent-opt, set your keys, and build the dataset this example needs
+
+
+ How optimizers, prompts, and runs fit together
+
+
+ A learning-based optimizer for few-shot prompt tuning
+
+
diff --git a/src/pages/docs/optimization/reference/optimizers/random-search.mdx b/src/pages/docs/optimization/reference/optimizers/random-search.mdx
new file mode 100644
index 00000000..0a31303c
--- /dev/null
+++ b/src/pages/docs/optimization/reference/optimizers/random-search.mdx
@@ -0,0 +1,95 @@
+---
+title: "Random Search"
+description: "Parameters, defaults, and a runnable example for the Random Search optimizer."
+---
+
+## When to use Random Search
+
+Random Search is the cheapest way to find out how much headroom a prompt has before reaching for a directed optimizer, one that uses each round's scores to steer the next (see [choosing an optimizer](/docs/optimization/concepts/choosing-an-optimizer) to compare it against the other five). A run returns the highest-scoring variation it found along with its score. It generates a fixed batch of independent variations of your starting prompt and scores each one against your dataset: no variation feeds into the next, so the score for variation 2 has no effect on what variation 3 looks like. Run it from the [platform UI](/docs/optimization/guides/run-an-optimization) or the Python SDK.
+
+## Parameters
+
+The **On-screen label** column is the field name shown when you run this optimizer from the UI; see [Run an optimization](/docs/optimization/guides/run-an-optimization) for the full form walkthrough. The **Default** column shows what applies when you don't set the value yourself: the UI form's prefilled value, or the SDK's fallback when the argument is omitted.
+
+| Parameter | On-screen label | Default | Description |
+|---|---|---|---|
+| `num_variations` | Number Variations | 3 prefilled in the UI, 5 in the SDK | Number of independent prompt variations to generate and score |
+
+More variations cover more of the prompt space but cost proportionally more generation and evaluation calls, since each one is scored independently. Start at 3 for a quick read on headroom.
+
+The table above covers only this optimizer's tuning knob. `evaluator`, `data_mapper`, and `dataset`, also passed in the example below, are shared by every optimizer's `optimize()` call and are covered in the [SDK reference](/docs/sdk/optimization).
+
+## Usage
+
+Requires `pip install agent-opt`, which provides the `fi.opt` modules imported below, plus an `FI_API_KEY` and `FI_SECRET_KEY` pair (see [API keys](/docs/admin-settings/api-keys) for where to get them). Pass them as environment variables or, as below, directly into `Evaluator`.
+
+```python
+from fi.opt.optimizers import RandomSearchOptimizer
+from fi.opt.generators import LiteLLMGenerator
+from fi.opt.datamappers import BasicDataMapper
+from fi.opt.base.evaluator import Evaluator
+
+# Generator holding the starting prompt
+generator = LiteLLMGenerator(
+ model="gpt-4o-mini",
+ prompt_template="Summarize this article: {article}"
+)
+
+# Evaluator that scores each variation
+evaluator = Evaluator(
+ eval_template="summary_quality",
+ eval_model_name="turing_flash",
+ fi_api_key="your_key",
+ fi_secret_key="your_secret"
+)
+
+# Maps generator output and dataset fields to what the evaluator expects.
+# "generated_output" is the generator's fixed output key; "article" is
+# the dataset field from this example and should match your own data.
+data_mapper = BasicDataMapper(
+ key_map={"input": "article", "output": "generated_output"}
+)
+
+# Dataset: a plain list of dicts, one per example the optimizer scores the prompt against
+my_dataset = [
+ {"article": "The James Webb Space Telescope has captured its clearest images yet of a distant exoplanet's atmosphere, revealing traces of carbon dioxide and methane."},
+ {"article": "Researchers have discovered a new enzyme that breaks down PET plastic at room temperature, far faster than any previously known enzyme."},
+]
+
+optimizer = RandomSearchOptimizer(
+ generator=generator,
+ num_variations=3
+)
+
+result = optimizer.optimize(
+ evaluator=evaluator,
+ data_mapper=data_mapper,
+ dataset=my_dataset
+)
+
+print(f"Final score: {result.final_score:.4f}")
+print(f"Best prompt:\n{result.best_generator.get_prompt_template()}")
+```
+
+`result.final_score` is the winning variation's score and `result.best_generator.get_prompt_template()` is its prompt text; see [reading the result](/docs/optimization/guides/optimize-from-the-sdk#read-the-result) for what these fields mean and how to use them.
+
+`summary_quality` and `turing_flash` are real built-in names; see [eval templates](/docs/evaluation/builtin) and [evaluator models](/docs/evaluation/concepts/evaluator-models) for the full lists.
+
+A successful run prints something like:
+
+```
+Final score: 0.8532
+Best prompt:
+Summarize this article in 2-3 sentences, covering the main finding and its significance.
+```
+
+## Keep exploring
+
+
+
+ How optimizers, prompts, and runs fit together
+
+
+ A learning-based optimizer for few-shot prompt tuning
+
+
diff --git a/src/pages/docs/optimization/reference/sdk-api.mdx b/src/pages/docs/optimization/reference/sdk-api.mdx
new file mode 100644
index 00000000..0f1018fd
--- /dev/null
+++ b/src/pages/docs/optimization/reference/sdk-api.mdx
@@ -0,0 +1,206 @@
+---
+title: "SDK & API"
+description: "Evaluator, BasicDataMapper, LiteLLMGenerator, optimize() arguments, return types, and EarlyStoppingConfig for the agent-opt library."
+---
+
+Install the library with `pip install agent-opt`; every import on this page comes from the `fi.opt` package it provides. For a full walkthrough that builds and runs an optimization end to end, see [Optimize from the SDK](/docs/optimization/guides/optimize-from-the-sdk); for the environment variables and authentication this page's examples assume, see [Install and authenticate](/docs/optimization/guides/optimize-from-the-sdk#install-and-authenticate).
+
+## The shared surface
+
+Every optimizer in `agent-opt` is built on the same four pieces:
+
+- **`Evaluator`** scores outputs
+- **`BasicDataMapper`** maps dataset fields to what the evaluator expects
+- **`LiteLLMGenerator`** holds the prompt being optimized
+- **`optimizer.optimize()`** runs the optimizer and returns an `OptimizationResult`
+
+This page is the reference for that shared surface. Constructor parameters specific to a single optimizer live on that optimizer's own reference page.
+
+## Evaluator
+
+`Evaluator` has two construction modes. Provide `metric` for local evaluation, or provide `eval_template` together with `eval_model_name` for evaluation on the Future AGI platform.
+
+Platform mode runs a pre-built Future AGI eval template with no custom code. Local mode uses a custom metric, such as a local LLM-as-a-judge or a rule-based heuristic. See [Choosing Evaluation Metrics for Prompt Optimization](/docs/cookbook/eval-metrics-optimization) for worked examples of both modes, including where a local metric instance like `my_metric` below comes from.
+
+| Argument | Type | Default | Mode | Description |
+|---|---|---|---|---|
+| `eval_template` | `str` | required (Platform) | Platform | Name of the Future AGI platform eval template to run |
+| `eval_model_name` | `str` | required (Platform) | Platform | Model the platform eval template runs under |
+| `fi_api_key` | `str` | `None` | Platform | Future AGI API key; falls back to the `FI_API_KEY` environment variable when omitted |
+| `fi_secret_key` | `str` | `None` | Platform | Future AGI secret key; falls back to the `FI_SECRET_KEY` environment variable when omitted |
+| `metric` | `BaseMetric` | required (Local) | Local | A local metric instance that performs the evaluation |
+| `provider` | `LiteLLMProvider` | `None` | Local | Optional, only used with a local LLM-as-judge metric; defaults from environment variables when omitted (see [Install and authenticate](/docs/optimization/guides/optimize-from-the-sdk#install-and-authenticate) for which ones) |
+
+```python
+from fi.opt.base.evaluator import Evaluator
+
+# Platform mode
+evaluator = Evaluator(
+ eval_template="summary_quality",
+ eval_model_name="turing_flash",
+ fi_api_key="your_key",
+ fi_secret_key="your_secret"
+)
+
+# Local mode
+evaluator = Evaluator(metric=my_metric)
+```
+
+## BasicDataMapper
+
+`BasicDataMapper` is the only exported data mapper.
+
+| Argument | Type | Default | Description |
+|---|---|---|---|
+| `key_map` | `dict[str, str]` | required | Maps each key the evaluator expects to the dataset column or generator-output key it should read from, as `{evaluator_key: source_key}` |
+
+```python
+from fi.opt.datamappers import BasicDataMapper
+
+data_mapper = BasicDataMapper(
+ key_map={
+ "input": "article", # dataset column "article" -> evaluator's "input"
+ "output": "generated_output" # generator's output -> evaluator's "output"
+ }
+)
+```
+
+## LiteLLMGenerator
+
+`LiteLLMGenerator` is the only exported generator.
+
+| Argument | Type | Default | Description |
+|---|---|---|---|
+| `model` | `str` | required | LiteLLM-formatted model identifier |
+| `prompt_template` | `str` | required | Prompt template the generator fills to produce outputs |
+
+```python
+from fi.opt.generators import LiteLLMGenerator
+
+generator = LiteLLMGenerator(
+ model="gpt-4o-mini",
+ prompt_template="Summarize this article: {article}"
+)
+```
+
+`model` is routed through LiteLLM, so it needs its provider's API key set as an environment variable, for example `OPENAI_API_KEY` for `gpt-4o-mini` above. See [Install and authenticate](/docs/optimization/guides/optimize-from-the-sdk#install-and-authenticate) for the pattern.
+
+## optimize()
+
+Every optimizer implements `optimize()` with the same core parameters below as explicit named arguments, not as `**kwargs`; a runnable instance follows the argument table. `optimizer` is an instance of one of the optimizer classes on the [Optimizers](/docs/optimization/reference/optimizers) reference page, built in that example as `RandomSearchOptimizer` from the `generator` above.
+
+```python
+optimizer.optimize(evaluator, data_mapper, dataset, initial_prompts, early_stopping=None, **kwargs) -> OptimizationResult
+```
+
+| Argument | Type | Default | Description |
+|---|---|---|---|
+| `evaluator` | `Evaluator` | required | The `Evaluator` instance that scores generated outputs |
+| `data_mapper` | `BasicDataMapper` | required | The `BasicDataMapper` instance that maps dataset and output keys |
+| `dataset` | `list[dict]` | required | The dataset to evaluate against |
+| `initial_prompts` | `list[str]` | required (not accepted by Random Search) | Starting prompt(s) the optimizer refines; see each optimizer's reference page |
+| `early_stopping` | `EarlyStoppingConfig` | `None` | Stops the run before it reaches its maximum iterations; see [EarlyStoppingConfig](#earlystoppingconfig) below |
+| `**kwargs` | | optional | Optimizer-specific keyword arguments, where the optimizer accepts them (see each optimizer's reference page); Meta-Prompt and GEPA accept no `**kwargs` passthrough at all, so anything beyond their own named parameters (`task_description`, `num_rounds`, and `eval_subset_size` for Meta-Prompt; `max_metric_calls` for GEPA) raises a `TypeError` |
+
+```python
+from fi.opt.optimizers import RandomSearchOptimizer
+
+dataset = [
+ {"article": "..."},
+ {"article": "..."}
+]
+
+optimizer = RandomSearchOptimizer(generator=generator, num_variations=3)
+
+result = optimizer.optimize(
+ evaluator=evaluator,
+ data_mapper=data_mapper,
+ dataset=dataset
+)
+```
+
+`evaluator` and `data_mapper` are the values built in the sections above; `optimizer` is built just above from `generator`. See [Optimize from the SDK](/docs/optimization/guides/optimize-from-the-sdk) for a full walkthrough that builds `dataset`.
+
+## EarlyStoppingConfig
+
+
+`EarlyStoppingConfig` is not part of the current `0.0.1` release of `agent-opt` on PyPI. This section documents an upcoming release; importing `fi.opt.utils.early_stopping` against `0.0.1` raises an `ImportError`.
+
+
+Pass an `EarlyStoppingConfig` instance as the `early_stopping` keyword argument to `optimize()` to stop a run before it reaches its maximum iterations. All fields are optional. Early stopping turns on when `patience`, `min_score_threshold`, or `max_evaluations` is set; `min_delta` alone does not enable it, it only tunes the patience counter. When more than one field is set, optimization stops as soon as any one of them is satisfied.
+
+| Field | Bounds | Description |
+|---|---|---|
+| `patience` | greater than 0 | Stop after this many consecutive iterations with no score improvement |
+| `min_score_threshold` | 0.0 to 1.0 | Stop once the score reaches or exceeds this threshold |
+| `max_evaluations` | greater than 0 | Stop once this many total dataset evaluations have run across all iterations; checked before the score threshold |
+| `min_delta` | 0.0 or greater | Minimum score improvement counted as progress |
+
+`optimize()` always returns an `OptimizationResult`, whether or not a stopping criterion triggered; check `early_stopped` and `stop_reason` on the result to see what happened.
+
+```python
+from fi.opt.utils.early_stopping import EarlyStoppingConfig
+
+result = optimizer.optimize(
+ evaluator=evaluator,
+ data_mapper=data_mapper,
+ dataset=dataset,
+ early_stopping=EarlyStoppingConfig(
+ patience=3,
+ min_score_threshold=0.9,
+ min_delta=0.01
+ )
+)
+```
+
+## Return values
+
+### OptimizationResult
+
+The object `optimize()` returns.
+
+| Field | Description |
+|---|---|
+| `best_generator` | The generator holding the best-performing prompt found |
+| `history` | List of `IterationHistory` records, one per iteration |
+| `final_score` | The best score achieved during the run |
+| `early_stopped` | Whether the run was terminated early by a stopping criterion |
+| `stop_reason` | Explanation for early stopping, when applicable |
+| `total_iterations` | Total number of iterations completed |
+| `total_evaluations` | Total number of dataset evaluations performed |
+
+```python
+print(result.final_score)
+print(result.best_generator.prompt_template)
+```
+
+### IterationHistory
+
+A single iteration's record inside `history`.
+
+| Field | Description |
+|---|---|
+| `prompt` | The prompt evaluated in this iteration |
+| `average_score` | Mean score across this iteration's evaluations |
+| `individual_results` | List of `EvaluationResult`, one per dataset row |
+
+### EvaluationResult
+
+A single evaluation's result, returned inside `individual_results`.
+
+| Field | Description |
+|---|---|
+| `score` | Normalized score, 0.0 to 1.0 |
+| `reason` | Explanation for the score |
+| `metadata` | Additional evaluator-specific metadata |
+
+## Keep exploring
+
+
+
+ Parameters and defaults for each of the six optimizers
+
+
+ How optimizers, prompts, and runs fit together
+
+
diff --git a/src/pages/docs/optimization/troubleshooting.mdx b/src/pages/docs/optimization/troubleshooting.mdx
new file mode 100644
index 00000000..d4b5ce0b
--- /dev/null
+++ b/src/pages/docs/optimization/troubleshooting.mdx
@@ -0,0 +1,61 @@
+---
+title: "Optimization FAQ & fixes"
+description: "Symptom-first fixes for optimization run and SDK errors"
+---
+
+## In this page
+
+This page covers common errors when starting, running, or reading an optimization, and how to fix them.
+
+## Common errors and fixes
+
+**Runs that won't start or won't stop**
+
+| Symptom | Cause | Fix |
+|---|---|---|
+| The **Run Optimization** drawer shows a **Run Prompt** button instead of the run fields | The dataset has no column of generated outputs yet, so there's nothing to optimize | [Run a prompt](/docs/dataset/guides/run-a-prompt-on-every-row) against the dataset, then reopen the drawer |
+| Starting the run is blocked with 'Add evaluations before starting your optimization run' | Evals are the run's objective; a run needs at least one to score against | Add an eval in the drawer's [evaluations section](/docs/optimization/guides/run-an-optimization#fill-the-run-drawer) before clicking **Start Optimization** |
+| The run is rejected with a 'Missing required keys for optimizer ...' or 'Unexpected keys provided for optimizer ...' error | Each optimizer accepts only its own exact set of parameters | Fill in only the fields the drawer shows for the optimizer you selected; see [Optimizers](/docs/optimization/reference/optimizers) for the exact parameter set per algorithm |
+| A self-hosted deployment returns HTTP 402 when starting a run | Optimization is a paid feature, gated separately from the rest of the platform | Optimization needs to be enabled on your deployment; contact support to have it turned on |
+| **Stop** isn't on the run's row | Stop only shows while a run is **Queue** or **Running** | There's nothing to stop; if the run finished as **Completed**, check the [trial list](/docs/optimization/guides/read-optimization-results#the-trial-list) for its results |
+
+**SDK**
+
+| Symptom | Cause | Fix |
+|---|---|---|
+| GEPA raises an import or missing-library error from the SDK | `gepa` is a required dependency of `agent-opt`, so this points to an incomplete or broken install | Run `pip install gepa`, as the SDK's ImportError message tells you; if that doesn't resolve it, reinstall `agent-opt` (`pip install --force-reinstall agent-opt`) |
+
+## Why a run failed or scores are flat
+
+### Failed before any trial ran
+
+The run couldn't be started, so it's marked **Failed** immediately with no trials to open. The message on the run's page is generic and just points you to contact support.
+
+### Failed partway through
+
+The run failed after it had started. The run's page shows a **Failed to optimize** panel with the error message from the point of failure; read it first to tell whether the run broke early, before any trial produced a usable score, or later on.
+
+If the message points to your dataset or eval setup, check that the column being evaluated has a value for every row, or run the eval on its own outside optimization to see whether it fails the same way. If it points elsewhere, rebuild the run in the drawer to try again.
+
+### Scores didn't improve across trials
+
+Check your eval selection and your sample size: a trial is only as good as what it's scored against, and every run samples at most 50 dataset rows, so a prompt that looks flat there may behave differently across the rest of your data. See [A run scores at most 50 rows, and evals decide what counts as better](/docs/optimization/concepts/understanding-optimization#a-run-scores-at-most-50-rows-and-evals-decide-what-counts-as-better) for how each one shapes the score.
+
+Still stuck? Reach out via [support](https://futureagi.com/contact-us).
+
+## Keep exploring
+
+
+
+ Fill the Run Optimization drawer and launch a run from the platform
+
+
+ Walk a run's score graph, trial list, and per-row scores
+
+
+ The feedback loop, key components, and how to choose an algorithm
+
+
+ Run the same kind of optimization from code with agent-opt
+
+
diff --git a/src/pages/docs/prompt/concepts/prompt-engineering.mdx b/src/pages/docs/prompt/concepts/prompt-engineering.mdx
index 1e4b034d..9842e559 100644
--- a/src/pages/docs/prompt/concepts/prompt-engineering.mdx
+++ b/src/pages/docs/prompt/concepts/prompt-engineering.mdx
@@ -1,59 +1,83 @@
---
-title: "Prompt Engineering: Crafting Effective Prompts in Future AGI"
-description: "What prompt engineering is, how to think about crafting effective prompts, and how the Prompt Workbench supports the iteration process."
+title: "Prompt Engineering"
+description: "A mental model for what makes a prompt reliable, and how to iterate when it isn't"
---
-## About
+## What prompt engineering is
-Prompt engineering is the practice of designing and refining the instructions you give a language model to get reliable, high-quality responses. Unlike traditional software where behavior is determined by code, a language model's behavior is largely shaped by the prompt — the wording, structure, context, and examples you provide directly influence what the model produces.
+**Prompt engineering** is getting a model to do what you actually meant, through the words, structure, and examples you give it, not through a setting you flip. The same model, on the same task, produces a noticeably better or worse result depending on how the prompt is written.
-In the Prompt Workbench, prompt engineering is a structured workflow: you write a prompt, test it against real inputs, evaluate the outputs, and iterate. The platform tracks every version, so you can measure whether a change improved results or regressed them, and roll back if needed.
+## Five levers that make a prompt work
----
+- **Explicit task**: state exactly what you want, not what you're avoiding. "Summarize this in three bullet points for a non-technical audience" beats "summarize this"
+- **System message**: split system and user messages. The system message sets role, tone, and constraints once, while the user message carries the input
+- **Output format**: state the output format you want. If you need JSON, a fixed length, or a specific structure, say so directly rather than implying it through example alone
+- **Few-shot examples**: show, don't just tell, for nuanced judgment. An assistant message is a valid part of a prompt, so one or two example responses written as assistant turns teach a style or a judgment call more reliably than a paragraph describing it
+- **Relevant context**: only include context the model needs. More context isn't automatically better; irrelevant material distracts the model and adds cost without adding accuracy
-## Principles of a good prompt
+Here's a weak prompt and the same task with all five levers applied:
-**Be explicit about the task.** A model performs better when the instruction is unambiguous. Instead of "summarize this," say "summarize this in three bullet points for a non-technical audience." The more specific the instruction, the less the model has to infer.
+**Weak prompt:**
-**Use the system message for behavior, the user message for input.** The system message sets the model's role, tone, and constraints. The user message carries the actual task or question. Keeping these separate makes it easier to reuse the same behavior across many different inputs.
+```
+Here's the customer's full message history, all 40 messages. Look at the most recent one and tell me how they feel about it.
+```
-**Provide output format requirements.** If you need JSON, a list, a specific length, or a particular structure, say so explicitly. Models follow formatting instructions well when they are clear and placed consistently in the prompt.
+**Same task, with all five levers applied:**
-**Use few-shot examples for complex tasks.** When the task involves nuanced judgment or a specific style, including one or two assistant message examples shows the model exactly what you expect. Examples are more reliable than lengthy descriptions of what "good" looks like.
+System message:
-**Keep context relevant.** More context is not always better. Irrelevant context can distract the model and increase cost. Include only what the model needs to complete the task.
+```
+You are a support assistant. Be concise and avoid jargon.
+```
----
+User message (example turn):
-## The iteration cycle
+```
+Classify the sentiment of this customer feedback as Positive, Negative, or Neutral, and give a one-word reason.
-Prompt engineering is iterative. A first draft rarely performs optimally across all inputs — the process is:
+Feedback: "The product arrived broken and support never replied."
+```
-1. **Write**: Draft a prompt with a clear task, role, and output format.
-2. **Test**: Run it against a representative set of real inputs, not just the easy cases.
-3. **Evaluate**: Score the outputs — manually or with an automated evaluator — to identify where the prompt fails.
-4. **Refine**: Change one thing at a time. Adjust wording, add an example, tighten the instruction, or change the model.
-5. **Compare**: Use version history to compare the new version against the previous one on the same inputs.
+Assistant message (example turn):
-Changing multiple things at once makes it hard to know what caused an improvement or regression. Small, targeted changes with consistent evaluation produce more reliable results.
+```
+Negative - defect
+```
----
+User message (real input):
-## Common failure modes
+```
+Feedback: "The app keeps crashing when I try to export my report."
+```
-| Failure | Likely cause |
-|---|---|
-| Inconsistent output format | Format not explicitly specified, or specified only in prose |
-| Model ignores part of the instruction | Instruction is buried, ambiguous, or contradicts itself |
-| Output too long or too short | Max tokens not set, or length guidance missing from prompt |
-| Model hallucinates facts | No grounding context provided, or no instruction to say "I don't know" |
-| Tone or style varies across runs | Persona or tone not defined in the system message |
+## From symptom to lever
----
+When a prompt's output goes wrong, it's tempting to guess at the cause. It's more useful to ask which lever above actually controls it, since most failures come down to exactly one of the five.
+
+| Symptom | Lever | Fix |
+|---|---|---|
+| Instruction gets partly ignored | Explicit task | State the instruction plainly instead of burying it in a longer prompt |
+| Tone drifts across runs | System message | Define persona and tone in the system message instead of leaving them implicit |
+| Format or length is inconsistent | Output format | Spell out the exact structure or length you want |
+| A judgment call misses the mark | Few-shot examples | Add one or two assistant message examples showing the call you want |
+| Model hallucinates or drifts off-topic | Relevant context | Give it only the context it needs, and tell it to say "I don't know" when that's not enough |
+
+## The iteration loop
+
+Prompt engineering rarely lands on the first try. [Run the prompt](/docs/prompt/guides/run-a-prompt) against real inputs, [evaluate the outputs](/docs/prompt/guides/evaluate-prompt-outputs) instead of eyeballing them, and [commit and compare versions](/docs/prompt/guides/commit-and-compare-versions) to check whether a change helped. Change one thing at a time, so if the score moves, you know which change caused it.
+
+
+This page covers the wording. Errors or disabled buttons in the product itself belong on [Prompt FAQ & fixes](/docs/prompt/troubleshooting).
+
-## Next steps
+## Keep exploring
-- [Understanding Prompts](/docs/prompt/concepts/understanding-prompts): Prompt structure, roles, and model configuration.
-- [Prompt Variables](/docs/prompt/concepts/understanding-prompts): How to use variables to make a single template reusable.
-- [Versions and Labels](/docs/prompt/concepts/versions-and-labels): How versioning supports the iteration cycle.
-- [Create a Prompt from Scratch](/docs/prompt/features/create-from-scratch): Build your first prompt in the Workbench.
+
+
+ See the prompt object itself: templates, messages, and variables
+
+
+ Errors and disabled buttons in the product, not wording
+
+
diff --git a/src/pages/docs/prompt/concepts/understanding-prompts.mdx b/src/pages/docs/prompt/concepts/understanding-prompts.mdx
index d1553240..dbabbf10 100644
--- a/src/pages/docs/prompt/concepts/understanding-prompts.mdx
+++ b/src/pages/docs/prompt/concepts/understanding-prompts.mdx
@@ -1,113 +1,82 @@
---
-title: "Understanding Prompts in Future AGI Prompt Workbench"
-description: "Explains what a prompt is, how it is structured, how variables work, and how prompts connect to models in the Prompt Workbench."
+title: "Understanding Prompts"
+description: "What a prompt template is made of, and how variables fill it in"
---
-## About
+## What a prompt template is
-A prompt is the instruction you send to a language model to produce a response. It tells the model who it is, what it should do, and what input to work with. Getting the prompt right is one of the most direct ways to improve the quality of your AI product.
+A **prompt template** is a named object saved inside a workspace. It holds three things: an ordered list of messages, a model configuration, and the variable names the messages reference.
-In the Prompt Workbench, prompts are managed as templates. A template is a saved, versioned prompt that can be reused across datasets, simulations, experiments, and your application via the SDK.
+Take `support-agent`, the template this page uses as its running example. It's built for a customer-support agent: a system message, a user message, a model configuration that picks the model and its settings, and the variables `{{company_name}}` and `{{customer_question}}` its messages fill in. Every template you build has this same shape, whatever it does.
----
-
-## Structure
+## The object model
-A prompt in Future AGI is made up of one or more messages, each with a role:
+A template doesn't stand alone.
-| Role | Purpose |
-|---|---|
-| **System** | Sets the model's behavior, persona, and constraints. Optional but highly effective for controlling tone and scope. |
-| **User** | The actual input or instruction sent to the model. This is where the task or question lives. |
-| **Assistant** | Used for few-shot examples: you provide sample responses to show the model the format or style you expect. |
+- It can be seeded from a **base template**, a reusable starter you begin from and edit into your own template instead of an empty editor. Base templates come from the product's built-in library or from prompts your team has saved as reusable starting points, and you pick one from the template browser when you [start a new prompt](/docs/prompt/guides/create-a-prompt#start-with-a-template)
+- It can live in a [folder](/docs/prompt/guides/organize-prompts-in-folders), so a workspace with a growing library stays navigable
+- It carries a version history, built up as you commit
-Most prompts have at least a system message and a user message. The system message shapes how the model behaves; the user message drives what it produces.
+ PT["Prompt template · support-agent"]
+ WS --> FL["Folder"]
+ FL -. optional .-> PT
+ BT["Base template"] -. seeds .-> PT
+ PT --> MSG["Messages (ordered) · system, user, assistant"]
+ PT --> CFG["Model configuration"]
+ PT --> VAR["Variable names · company_name, customer_question"]
+ PT -- commit --> PV["Version · snapshot"]`}
+/>
----
-
-## Variables
+Two of these edges are easy to miss. Seeding from a base template only sets the starting content: once `support-agent` exists, editing it never touches the base template it began from. And committing produces a version, a snapshot of the messages, model configuration, and variables at that point. This page stops at that fact; how [versions and labels](/docs/prompt/concepts/versions-and-labels) work together is its own page.
-Variables make a prompt template reusable. Instead of hardcoding specific values, you use placeholders that get replaced with real data at runtime. This lets a single template run against many different inputs without being rewritten.
+## Messages and roles
-### Syntax
+The messages inside a template are ordered, and each one carries exactly one of three roles: `system`, `user`, and `assistant`. A message with any other role is rejected, not just discouraged by the editor.
-Variables use double curly brace syntax: `{{variable_name}}`. You can place them anywhere in the system or user message content.
+In `support-agent`, the system message sets the agent's behavior ("You are a support agent for `{{company_name}}`") and the user message carries the incoming question (`{{customer_question}}`).
-```
-You are a support agent for {{company_name}}.
-
-Answer the following customer question clearly and professionally:
-{{customer_question}}
-```
+## Model configuration
-When this prompt is run, `{{company_name}}` and `{{customer_question}}` are replaced with the actual values you supply.
+The model configuration covers three decisions:
-### How variables are supplied
+- **Which model** runs the template
+- **Generation settings**, things like temperature and max tokens
+- **What the model is allowed to produce**: a plain string, a tool call, or a structured response matching a saved schema
-**In the UI**: When you run a prompt against a dataset, you map dataset columns to the variable names the template expects.
+It's saved on the template alongside the messages and variables. See [Model configuration](/docs/prompt/reference/model-configuration) for the full field list and validation rules.
-**In the SDK**: You pass a dictionary of variable names and values to the `compile()` method:
+## Variables
-```python
-compiled = client.compile(
- company_name="Acme Corp",
- customer_question="How do I reset my password?"
-)
-```
+Variables are `{{name}}` markers inside a message's content, substituted with real values at run time. `support-agent`'s user message reads something like:
-```typescript
-const compiled = client.compile({
- company_name: "Acme Corp",
- customer_question: "How do I reset my password?"
-});
```
-
-The `compile()` method returns the fully resolved messages, ready to send to a model.
-
-### Placeholder messages
-
-For dynamic chat history or multi-turn conversations, you can use a placeholder message instead of a variable inside a string. A placeholder is a special message with `type: "placeholder"` and a name. At compile time, you supply an array of messages for that key, and they are inserted into the message list at that position.
-
-```python
-tpl = PromptTemplate(
- name="chat-template",
- messages=[
- SystemMessage(content="You are a helpful assistant."),
- {"type": "placeholder", "name": "history"},
- UserMessage(content="{{question}}"),
- ],
-)
-
-compiled = client.compile(
- question="What is the refund policy?",
- history=[
- {"role": "user", "content": "Hi"},
- {"role": "assistant", "content": "Hello! How can I help?"}
- ]
-)
+Answer the following customer question clearly and professionally:
+{{customer_question}}
```
-This is useful when your prompt needs to include prior conversation turns that are only known at runtime.
-
----
+The template stores the variable names it expects, `company_name` and `customer_question`, so the platform knows what to ask for wherever the template runs. Where the values themselves come from depends on where you run it: [typed into the editor](/docs/prompt/guides/run-a-prompt#declare-variables-and-supply-values), [pulled from a dataset's columns](/docs/dataset/guides/run-a-prompt-on-every-row), or [passed in from your application](/docs/prompt/reference/sdk-api#compile).
-## Model Configuration
+## Placeholder messages
-Each prompt template includes a model configuration: the model to use and the parameters that control its output.
+A placeholder message is a different thing from a variable, and it's easy to confuse the two. A variable substitutes a value inside a message's content, so the message stays a single string. A placeholder is tracked separately from a template's variables, as its own named entry on the template, rather than as a substitution inside a message's content.
-| Setting | What it controls |
-|---|---|
-| **Model** | Which LLM processes the prompt |
-| **Temperature** | Randomness of the output. Higher values produce more varied responses. |
-| **Max Tokens** | Maximum length of the response |
-| **Top P** | Token selection diversity |
-| **Presence / Frequency Penalty** | Controls repetition in the output |
-| **Response Format** | Output format, e.g. plain text or JSON |
+## Why the object model matters
----
+Every piece above, from the model configuration down to placeholders, lives on the template itself. That's why `support-agent` runs the same wherever you call it, and why a version always tells you exactly what ran.
-## Next Steps
+## Keep exploring
-- [Versions and Labels](/docs/prompt/concepts/versions-and-labels): How prompt versioning and deployment labels work.
-- [Create a Prompt from Scratch](/docs/prompt/features/create-from-scratch): Build your first prompt in the Workbench.
-- [Prompt SDK](/docs/prompt/features/sdk): Full SDK reference for compiling and fetching prompt templates.
+
+
+ How a draft becomes a version, and how labels point at one
+
+
+ Writing messages that get a better result out of the model
+
+
+ Open an editor on a new template
+
+
diff --git a/src/pages/docs/prompt/concepts/versions-and-labels.mdx b/src/pages/docs/prompt/concepts/versions-and-labels.mdx
index cd04695a..f41298e8 100644
--- a/src/pages/docs/prompt/concepts/versions-and-labels.mdx
+++ b/src/pages/docs/prompt/concepts/versions-and-labels.mdx
@@ -1,69 +1,66 @@
---
-title: "Versions and Labels: Prompt Versioning in Future AGI"
-description: "Explains how prompt versioning and deployment labels work in the Prompt Workbench and how to manage multiple versions in Future AGI."
+title: "Versions & Labels"
+description: "How draft edits become versions, and how labels decide which one is live"
---
-## About
+## A version is a snapshot, a label is a pointer
-Every time you save a change to a prompt template, a new version is created. Versions give you a full history of how the prompt has changed over time, so you can compare iterations, understand what was changed, and roll back to any previous state if a change regresses quality.
-
-Labels sit on top of versions and control which version is active in each environment. Instead of hardcoding a specific version in your application, you fetch by label. When you want to promote a new version to production, you reassign the label — no application code change needed.
-
----
+A [prompt template](/docs/prompt/concepts/understanding-prompts) accumulates two different kinds of object as you work on it. A **version** is a saved snapshot of the template, created the moment you commit: the action that freezes your current draft into a permanent, numbered step. A **label** is a named pointer that sits on top of one version and marks it as the one you actually want live right now. Promoting and rolling back both come down to reassigning a label rather than editing the template again, which is why your application can fetch by label instead of hardcoding a version name.
## Versions
-Each version represents a snapshot of the prompt at a point in time. Versions are created when you commit a draft.
-
-The version lifecycle:
-1. **Draft**: An in-progress edit that has not been committed yet. Drafts are not fetchable by label.
-2. **Committed**: A saved, immutable version. Can be assigned a label and fetched by name and version number.
-3. **Default**: One version can be marked as the default, used when no version or label is specified.
-
-You can compare any two committed versions side by side in the Workbench and roll back to any previous version at any time.
-
----
-
-## Labels
-
-Labels are named pointers to specific versions. They let you control which version your application uses without changing code.
-
-| Label | Typical use |
-|---|---|
-| **Production** | The version your live application uses |
-| **Staging** | A candidate version being tested before promotion |
-| **Development** | Work in progress for active development |
-| **Custom labels** | Any label you create for your own deployment workflow |
+Until you commit, your edits are a draft: an in-progress, uncommitted state that isn't a version yet. Each version:
-A label always points to exactly one version. When you reassign a label to a new version, all applications fetching by that label immediately get the new version on their next request.
+- Gets a **name** the platform assigns automatically, `v1`, `v2`, `v3` and so on, matching the pattern `^v\d+$`. You don't type a version name yourself
+- Is **unique within its template**: the same template can't have two versions sharing a name
+- Can be marked the template's **default**, the version a fetch falls back to when nothing more specific points at one. Only one version is the default at a time
----
-
-## Fetching by label
-
-In your application, fetch a prompt by name and label instead of by version number. This decouples your code from specific versions:
+Saving a new version as default automatically takes it off whichever version had it before, without you touching that older version directly.
-```python
-# Always gets whichever version is labeled Production
-prompt = Prompt.get_template_by_name("my-template", label="Production")
-```
+
+There's no way to delete a single version on its own. Deleting a template deletes every version under it, not just the one you're looking at. There's no undo once the template itself is gone.
+
-```typescript
-const prompt = await Prompt.getTemplateByName("my-template", { label: "Production" });
-```
-
-To promote a new version, reassign the label in the Workbench or via the SDK. Your application picks it up automatically.
-
----
+To commit a draft and set a default, see [Commit & compare versions](/docs/prompt/guides/commit-and-compare-versions).
-## Rollback
-
-To roll back to a previous version, assign the label back to that version. The application immediately starts using the previous version on its next fetch. No redeployment needed.
-
----
-
-## Next steps
+## Labels
-- [Understanding Prompts](/docs/prompt/concepts/understanding-prompts): Prompt structure and components.
-- [Prompt Variables](/docs/prompt/concepts/understanding-prompts): How variables and compile work.
-- [Prompt SDK](/docs/prompt/features/sdk): Full SDK reference for versioning, labels, and fetching.
+A label is a pointer, not a copy. It's defined once for your workspace and reused across every template there. On any one template it sits on exactly one version at a time, and assigning it to a different version removes it from whichever version it was on before.
+
+- **System labels**: **Production**, **Staging**, and **Development** are created automatically and available in every workspace. The names are reserved in every workspace too, so nobody can create a second label called Production (or Staging, or Development)
+- **Custom labels**: create your own, one per region or per customer tier for example. The name is claimed once across your organization and workspace
+
+For fetching and reassigning labels from your own code, see the [SDK & API reference](/docs/prompt/reference/sdk-api).
+
+## Promoting and rolling back
+
+|commit| V2["v2"] -->|commit| V3["v3"]
+ end
+ Dr["Draft · in-progress edit"] -->|commit| V3
+ subgraph LABELS["Labels"]
+ Prod["Production"]
+ Stg["Staging"]
+ Dev["Development"]
+ end
+ Prod -.-> V2
+ Stg -.-> V3
+ Dev -.-> V1`} />
+
+The chain only grows one commit at a time.
+
+Say you edit the support-agent template's system message. That's a draft, invisible to anything reading by label. You commit it and it becomes v4. You point Staging at v4 to try it out, and once it looks good you move Production there too. A week later it regresses, so you point Production back at v2. The chain still has v4 in it, you've just moved a pointer.
+
+## Keep exploring
+
+
+
+ Turn a draft into a version, and compare a few side by side
+
+
+ Fetch and assign versions and labels from your own code
+
+
diff --git a/src/pages/docs/prompt/features/create-from-scratch.mdx b/src/pages/docs/prompt/features/create-from-scratch.mdx
deleted file mode 100644
index 7071edea..00000000
--- a/src/pages/docs/prompt/features/create-from-scratch.mdx
+++ /dev/null
@@ -1,118 +0,0 @@
----
-title: "Create Prompt from Scratch in Future AGI Prompt Workbench"
-description: "Build a new prompt manually in the Prompt Workbench with full control over structure, model, parameters, and variables."
----
-
-## About
-
-A prompt is made up of instructions, context, and variables that tell a model what to do and how to respond. When you build one from scratch, you control every part of it: the system instruction that shapes the model's behavior, the user message that drives the response, and the variables that make inputs dynamic at runtime.
-
-Use this when you have a clear idea of what the prompt should do, need precise control over structure and parameters, or are working on a use case that no existing template covers.
-
----
-
-## When to use
-
-- **Domain-specific prompts**: Your use case doesn't fit any existing template and requires rules or structure specific to your domain.
-- **Precise output formatting**: You need the model to return a specific schema or structured format and want full control over how the instruction is written.
-- **Agent prompts with custom tools**: You are building an agent that calls your own APIs and want to define the tools and system instruction together from the start.
-- **Bringing an existing prompt into the Workbench**: You have prompt text already written elsewhere and want to move it into the Workbench for versioning and testing.
-
----
-
-## How to
-
-
-
- From the Future AGI dashboard, locate the navigation panel on the left side of the screen. Under the "Build" section, click on "Prompts" to access the prompts management interface.
-
- 
-
-
-
- Once in the Prompts section, click on the "Create prompt" button located on the right side of the screen. This will open a modal dialog with prompt creation options.
-
- 
-
- In the "Create a new prompt" modal, you have three options:
- - **Generate with AI**: Automatically generate a prompt using AI
- - **Start from scratch**: Create a prompt manually
- - **Start with a template**: Use a pre-made template
-
- For this guide, select "Start from scratch" to create your prompt manually.
-
- 
-
-
-
- Now you'll be taken to the prompt editor interface where you can configure various aspects of your prompt:
-
- - **Rename your prompt**: By default, your prompt will be named "Untitled-1". To rename it, click on the title and enter a more descriptive name that reflects the purpose of the prompt.
-
- 
-
- - **Choose a model**: Click on "Select Model" to choose which AI model you want to use for your prompt. Future AGI offers various models with different capabilities.
-
- 
-
- - **Configure model parameters**: After selecting a model, you can adjust its parameters to fine-tune the AI's behavior:
- - **Temperature**: Controls randomness (higher values = more creative, lower values = more deterministic)
- - **Top P**: Influences token selection diversity
- - **Max Tokens**: Sets the maximum length of the response
- - **Presence Penalty**: Reduces repetition by penalizing tokens based on their presence
- - **Frequency Penalty**: Reduces repetition by penalizing tokens based on their frequency
- - **Response Format**: Choose the output format (e.g. Text)
-
- Adjust these parameters to get the desired behavior from your AI model.
-
- 
-
- - **Add tools (optional)**: You can enhance your prompt by adding tools that give the AI additional capabilities. To add tools:
- - Click on the "Tools" tab in the right panel
- - Click "Create tool" to add a new tool
- - Configure the tool with a name, description, and input schema
-
- Tools allow your prompt to perform specific actions or access external data sources.
-
- 
-
-
-
- In the prompt editor, you'll see two main text areas:
-
- - **System (optional)**: Here you can provide system-level instructions that guide the overall behavior of the AI.
- - **User**: This is where you write the actual prompt that will be presented to the AI.
-
- Write your prompt in the appropriate fields. Make it clear, specific, and include any necessary context or examples.
-
- When you're satisfied with your prompt, click the "Run Prompt" button in the top-right corner to execute it and see the AI's response.
-
- 
-
-
-
- After running your prompt, you can:
- - **Save it as a template**: Save the prompt as a template for future use so you or your team can reuse it.
- - **Iterate and refine**: Adjust the prompt or model parameters based on the responses you receive, then run again.
- - **Create variations**: Duplicate the prompt and try different wording or settings to compare approaches.
-
-
-
----
-
-## Next Steps
-
-
-
- Start from a pre-built template instead of scratch.
-
-
- Generate a prompt from a plain-language description.
-
-
- Connect prompts to traces to monitor performance in production.
-
-
- Fetch and use prompts programmatically from your application.
-
-
diff --git a/src/pages/docs/prompt/features/create-from-template.mdx b/src/pages/docs/prompt/features/create-from-template.mdx
deleted file mode 100644
index 0e1985e1..00000000
--- a/src/pages/docs/prompt/features/create-from-template.mdx
+++ /dev/null
@@ -1,105 +0,0 @@
----
-title: "Create Prompt from Existing Template in Future AGI"
-description: "Start from a pre-built prompt template in the Future AGI Prompt Workbench and customize it for your specific use case and model."
----
-
-## About
-
-Templates are pre-built prompts for common tasks. Instead of writing from a blank page, you start from a structure that already works, replace the placeholders with your context, and run it.
-
-The Prompt Workbench includes templates for summarization, customer support, analytics, content generation, and more. Each one comes with the system instruction, user message, and variables pre-filled. You customize what you need and leave the rest.
-
----
-
-## When to use
-
-- **Your task fits a common pattern**: Your use case is summarization, customer support, Q&A, analytics, or similar and you don't want to figure out the prompt structure yourself.
-- **Onboarding a team**: You want everyone starting from the same proven base so prompts are consistent across your team.
-- **Exploring what's possible**: You want to see working examples of well-structured prompts before building your own.
-- **Getting something running quickly**: You need a working prompt now and plan to refine it from there rather than start from a blank page.
-
----
-
-## How to
-
-
-
- From the Future AGI dashboard, locate the navigation panel on the left side of the screen. Under the "Build" section, click on "Prompts" to access the prompts management interface.
-
- 
-
-
-
- Once in the Prompts section, click on the "Create prompt" button located on the right side of the screen.
-
- 
-
- In the "Create a new prompt" modal, you'll see three options. Select "Start with a template" to browse available templates.
-
- 
-
-
-
- The template browser will open, showing different categories of templates on the left sidebar and available templates on the right.
-
- - Browse templates by category using the sidebar navigation
- - Search for specific templates using the search bar at the top
- - Click on a template card to view more details about it
-
- 
-
- When you find a template that matches your needs, review its description and purpose. Templates are pre-configured prompts designed for specific use cases like summarization, analytics, support, and more.
-
-
-
- After selecting a template, click the "Use this template" button in the top-right corner to create your prompt based on the template.
-
- The prompt editor will open with pre-filled content from the selected template. The system and user message fields will contain expert-crafted prompts that you can use as-is or modify.
-
- 
-
-
-
- Templates often include variables in `{{BRACKETS}}` or other formatting that you should replace with your specific information:
-
- - Review the system prompt and update any placeholders with your specific context
- - Modify the user message as needed for your particular use case
- - Adjust model parameters if necessary (temperature, tokens, etc.)
-
- 
-
- Many templates include helpful comments explaining how to use them effectively. Pay attention to these instructions to get the best results.
-
-
-
- Once you've customized the template to your needs, click the "Run Prompt" button in the top-right corner to execute it and see the AI's response.
-
- Review the output to ensure it meets your requirements. You may need to iterate on your customizations to get the exact results you're looking for.
-
-
-
- After running your prompt, you can:
- - **Save your customized version as a new template**: Save it for future use so you or your team can start from this version.
- - **Make further refinements**: Tweak the prompt or model parameters based on the responses you receive, then run again.
- - **Explore other templates**: Try different templates to discover effective prompt patterns for other use cases.
-
-
-
----
-
-## Next Steps
-
-
-
- Build a prompt manually with full control over structure and parameters.
-
-
- Generate a prompt from a plain-language description.
-
-
- Connect prompts to traces to monitor performance in production.
-
-
- Fetch and use prompts programmatically from your application.
-
-
diff --git a/src/pages/docs/prompt/features/create-with-ai.mdx b/src/pages/docs/prompt/features/create-with-ai.mdx
deleted file mode 100644
index 31759564..00000000
--- a/src/pages/docs/prompt/features/create-with-ai.mdx
+++ /dev/null
@@ -1,79 +0,0 @@
----
-title: "Create Prompt with AI Using Future AGI's Generate Feature"
-description: "Generate a new prompt from a plain-language description using the Generate with AI feature in the Future AGI Prompt Workbench."
----
-
-## About
-
-Prompt engineering requires translating a goal into precise instructions a model can follow. Generate with AI removes the initial translation step: you describe what you want the prompt to do, and the platform generates the system instruction and user message for you.
-
-The output lands directly in the Prompt Workbench editor, where you can inspect the structure, edit any part of it, add variables, choose a model, and run it. You stay in full control of the final prompt: the AI gives you a starting point, not a finished product.
-
----
-
-## When to use
-
-- **Unfamiliar task type**: You know the outcome you want but are not sure how to structure the prompt instruction for it.
-- **Getting a first draft fast**: A generated draft is a faster starting point than a blank editor, even if you plan to heavily edit it.
-- **Exploring prompt approaches**: Generate multiple versions from different descriptions and compare which structure produces better results.
-- **Rapid prototyping**: When you need something testable quickly and will iterate based on the model's output.
-
----
-
-## How to
-
-
-
- From the Future AGI dashboard, locate the navigation panel on the left. Under **Build**, click **Prompts** to open the prompts management interface.
- 
-
-
-
- In the Prompts section, click **Create prompt** on the right. In the "Create a new prompt" modal, select **Generate with AI** (instead of "Start from scratch" or "Start with a template").
- 
-
-
-
- Describe what you want the prompt to do in a short statement. Be specific about the task, tone, or format you need.
- 
-
-
-
- The platform generates the prompt based on your statement. When it finishes, the prompt editor opens with the generated system and user content filled in.
- 
-
-
-
- In the prompt editor you can:
- - **Rename** the prompt and **choose a model** and parameters if needed.
- - **Edit** the generated system and user text, add variables in `{{brackets}}`, or adjust formatting.
- - **Run Prompt** to test and see the model's response.
-
- 
-
-
-
- After generating your prompt, you can:
- - **Save it as a template**: Reuse the prompt or a tuned version as a template for your team.
- - **Iterate**: Change the statement and regenerate to try different drafts.
-
-
-
----
-
-## Next Steps
-
-
-
- Build a prompt manually with full control over structure and parameters.
-
-
- Start from a pre-built template for common use cases.
-
-
- Connect prompts to traces to monitor performance in production.
-
-
- Fetch and use prompts programmatically from your application.
-
-
diff --git a/src/pages/docs/prompt/features/folders.mdx b/src/pages/docs/prompt/features/folders.mdx
deleted file mode 100644
index 0716ce64..00000000
--- a/src/pages/docs/prompt/features/folders.mdx
+++ /dev/null
@@ -1,63 +0,0 @@
----
-title: "Manage Prompt Folders to Organize Your Prompt Library"
-description: "Organize prompt templates into folders in the Future AGI Prompt Workbench to keep your workspace navigable as your library grows."
----
-
-## About
-
-Folders in the Prompt Workbench let you group and organize prompt templates so your library stays navigable as it grows. Instead of a flat list, you can structure prompts by team, project, task type, or any convention that fits your workflow.
-
-You create folders in the sidebar and move prompts into them at any time, or create new prompts directly inside a folder.
-
----
-
-## When to use
-
-- **Multiple teams or projects**: Each team manages their own prompts without their workspace getting mixed up with others.
-- **Grouping by task type**: Keep summarization, support, analytics, and other prompt types separate so they are easy to find.
-- **Onboarding new teammates**: A clear folder structure tells new team members where to find existing prompts and where to add new ones.
-
----
-
-## How to
-
-
-
- In the Prompts section, click **New folder** in the sidebar.
- 
-
-
- Type a name for the folder and confirm. The platform navigates you into the new folder automatically. Folder names must be unique within your workspace.
- 
-
-
- Right-click any prompt in the list and select **Move**. A modal shows the prompt's current location and a dropdown to pick the destination folder. Select the folder and confirm — the prompt moves immediately.
- 
-
-
- Right-click (or open the three-dot menu) on any folder and select **Rename**. Enter the new name and save. The name must be non-empty and unique within your workspace.
- 
-
-
- Right-click a folder and select **Delete**. Confirm in the dialog. Deleting a folder also soft-deletes all prompts and prompt versions inside it, so make sure you no longer need them before proceeding.
-
-
-
----
-
-## Next Steps
-
-
-
- Build a new prompt and save it in a folder.
-
-
- Generate a prompt draft and organize it in a folder.
-
-
- Connect prompts to traces for metrics and monitoring.
-
-
- Manage and fetch prompts programmatically.
-
-
diff --git a/src/pages/docs/prompt/features/linked-traces.mdx b/src/pages/docs/prompt/features/linked-traces.mdx
deleted file mode 100644
index 10e84ba6..00000000
--- a/src/pages/docs/prompt/features/linked-traces.mdx
+++ /dev/null
@@ -1,76 +0,0 @@
----
-title: "Linked Traces: Monitor Prompt Performance in Production"
-description: "Associate prompts with production traces to monitor latency, token usage, and cost per prompt version in the Prompt Workbench."
----
-
-## About
-
-Every time your application sends a prompt to a model, Future AGI records it as a trace: the inputs, outputs, latency, tokens used, and cost. On their own, those traces tell you how your application is performing. Linked traces connect each trace back to the specific prompt and version that produced it.
-
-Once linked, the Prompt Workbench shows aggregated metrics per prompt version alongside the prompt itself. Instead of searching through individual traces, you see a consolidated view: how many times a prompt was called, its typical latency and cost, and how those metrics shift as you iterate.
-
-
-
----
-
-## When to use
-
-- **Validating a prompt change in production**: Compare latency and cost between versions on real traffic, not just test runs.
-- **Diagnosing a cost spike**: Metrics per prompt version show exactly which prompt or version is driving spend.
-- **Comparing active versions**: See real-world performance across prompt versions side by side to decide which to keep.
-- **Auditing prompt usage**: Trace count shows which prompts are actively being called and which are stale or abandoned.
-
----
-
-## Linked Traces vs Raw Traces
-
-| | Raw traces | Linked traces |
-|---|---|---|
-| **What you see** | Application-level metrics | Metrics per prompt and version |
-| **Attribution** | Anonymous API calls | Tied to a specific template and version |
-| **Where to view** | Observe / tracing dashboard | Prompt Workbench Metrics tab |
-| **Setup required** | SDK instrumentation | SDK instrumentation + template reference in request |
-
----
-
-## How to
-
-To link prompts to traces, you need to associate the prompt used in a generation with the corresponding trace. The process is described in the observability and manual tracing docs: [Log prompt templates](/docs/sdk/tracing/log-prompt-templates). Once your application sends traces that include the prompt template (or template ID), Future AGI links those traces to the prompt in the Prompt Workbench.
-
----
-
-## Metrics and Analytics
-
-After linking, open your prompt in the dashboard and go to the **Metrics** tab.
-
-| Metric | What it tells you |
-|---|---|
-| **Median Latency** | Typical time for the model to produce a response. Lower is better for responsiveness; use it to spot slow prompts or model changes. |
-| **Median Input Tokens** | Typical size of the prompt sent to the model. Helps you see verbosity and compare input length across versions. |
-| **Median Output Tokens** | Typical length of the model's reply. Useful for cost and length control; compare after changing instructions or max tokens. |
-| **Median Costs** | Typical cost per generation for this prompt. Use it to compare cost across prompt versions or models. |
-| **Traces Count** | How many times this prompt was used in the selected period. Shows which prompts are active and where to focus optimization. |
-| **First and Last Generation** | When the prompt was first and last used. Confirms the time range of the data you're viewing. |
-
-Compare the same metric across **prompt versions** or **time ranges** to see if a change improved latency, cost, or token usage.
-
----
-
-## Next Steps
-
-
-
- How versioning and deployment labels work.
-
-
- Manage and fetch prompts programmatically.
-
-
- Set up the trace-to-prompt connection in your application.
-
-
diff --git a/src/pages/docs/prompt/features/sdk.mdx b/src/pages/docs/prompt/features/sdk.mdx
deleted file mode 100644
index 9b947930..00000000
--- a/src/pages/docs/prompt/features/sdk.mdx
+++ /dev/null
@@ -1,384 +0,0 @@
----
-title: "Prompt Workbench SDK: Create and Version Prompts in Code"
-description: "Create, version, and run prompt templates programmatically using the Future AGI SDK for TypeScript/JavaScript or Python applications."
----
-
-## About
-
-The Prompt Workbench SDK lets you manage prompt templates programmatically. Instead of using the UI, you define, version, and deploy prompts from code using Python or TypeScript/JavaScript.
-
-This decouples prompt changes from application deploys. Your application fetches the active prompt by name and label at runtime, so you can update it on the platform without touching or redeploying your code. You can also assign labels like Production and Staging to control which version is live, run A/B tests across variants, and compile runtime variables into messages before sending them to a model.
-
----
-
-## When to use
-
-- **Prompts as part of CI/CD**: You want to version, commit, and deploy prompt changes through the same pipeline as your application code.
-- **Runtime prompt resolution**: Your application fetches the active prompt by name and label at runtime, so you can update prompts on the platform without a code deploy.
-- **A/B testing prompt variants**: You run multiple labeled versions of the same prompt in production and compare results across variants.
-- **Dynamic inputs at compile time**: Your prompts include placeholders for chat history or other message lists that are injected at runtime.
-
----
-
-## Installation
-
-
-
-```bash
-npm install @future-agi/sdk
-```
-
-```bash
-pip install futureagi
-```
-
-
-
-
-The Python package is installed as **`futureagi`** but imported as **`fi`** (e.g. `from fi.prompt.client import Prompt`).
-
-
----
-
-## Template structure
-
-### Basic components
-
-- **Name**: unique identifier (required)
-- **Messages**: ordered list of messages
-- **Model configuration**: model + generation params
-- **Variables**: dynamic placeholders used in messages
-
-### Message types
-
-- **System**: sets behavior/context
-- **User**: contains the prompt; supports variables like `{{var}}`
-- **Assistant**: few-shot examples or expected outputs
-
-```json
-{ "role": "system", "content": "You are a helpful assistant." }
-{ "role": "user", "content": "Introduce {{name}} from {{city}}." }
-{ "role": "assistant", "content": "Meet Ada from Berlin!" }
-```
-
----
-
-## Model configuration fields
-
-`model_name`, `temperature`, `frequency_penalty`, `presence_penalty`, `max_tokens`, `top_p`, `response_format`, `tool_choice`, `tools`
-
----
-
-## Placeholders and compile
-
-Add a placeholder message (`type="placeholder"`, `name="..."`) in your template. At compile time, supply an array of messages for that key; `{{var}}` variables are substituted in all message contents.
-
-
-
-```typescript JS/TS
-import { PromptTemplate, ModelConfig, MessageBase, Prompt } from "@future-agi/sdk";
-
-const tpl = new PromptTemplate({
- name: "chat-template",
- messages: [
- { role: "system", content: "You are a helpful assistant." } as MessageBase,
- { role: "user", content: "Hello {{name}}!" } as MessageBase,
- { type: "placeholder", name: "history" } as any, // placeholder
- ],
- model_configuration: new ModelConfig({ model_name: "gpt-4o-mini" }),
-});
-
-const client = new Prompt(tpl);
-// Compile with substitution and inlined chat history
-const compiled = client.compile({
- name: "Alice",
- history: [{ role: "user", content: "Ping {{name}}" }],
-} as any);
-```
-
-```python Python
-from fi.prompt import Prompt, PromptTemplate, ModelConfig, SystemMessage, UserMessage
-
-tpl = PromptTemplate(
- name="chat-template",
- messages=[
- SystemMessage(content="You are a helpful assistant."),
- UserMessage(content="Hello {{name}}!"),
- {"type": "placeholder", "name": "history"},
- ],
- model_configuration=ModelConfig(model_name="gpt-4o-mini"),
-)
-
-client = Prompt(template=tpl)
-compiled = client.compile(name="Alice", history=[{"role": "user", "content": "Ping {{name}}"}])
-```
-
-
-
----
-
-## Create templates
-
-
-
-```typescript JS/TS
-import { Prompt, PromptTemplate, ModelConfig, MessageBase } from "@future-agi/sdk";
-
-const tpl = new PromptTemplate({
- name: "intro-template",
- messages: [
- { role: "system", content: "You are a helpful assistant." } as MessageBase,
- { role: "user", content: "Introduce {{name}} from {{city}}." } as MessageBase,
- ],
- variable_names: { name: ["Ada"], city: ["Berlin"] },
- model_configuration: new ModelConfig({ model_name: "gpt-4o-mini" }),
-});
-
-const client = new Prompt(tpl);
-await client.open(); // draft v1
-await client.commitCurrentVersion("Finish v1", true); // set default
-```
-
-```python Python
-from fi.prompt import Prompt, PromptTemplate, ModelConfig, SystemMessage, UserMessage
-
-tpl = PromptTemplate(
- name="intro-template",
- messages=[
- SystemMessage(content="You are a helpful assistant."),
- UserMessage(content="Introduce {{name}} from {{city}}."),
- ],
- variable_names={"name": ["Ada"], "city": ["Berlin"]},
- model_configuration=ModelConfig(model_name="gpt-4o-mini"),
-)
-
-client = Prompt(template=tpl).create() # draft v1
-client.commit_current_version(message="Finish v1", set_default=True)
-```
-
-
-
----
-
-## Versioning (step-by-step)
-
-- Build the template (see above)
-- Create draft v1 (JS/TS: `await client.open()`; Python: `client.create()`)
-- Update draft & save (JS/TS: `saveCurrentDraft()`; Python: `save_current_draft()`)
-- Commit v1 and set default (JS/TS: `commitCurrentVersion("msg", true)`; Python: `commit_current_version`)
-- Open a new draft (JS/TS: `createNewVersion()`; Python: `create_new_version()`)
-- Delete if needed (JS/TS: `delete()`; Python: `delete()`)
-
----
-
-## Labels (deployment control)
-
-- **System labels**: Production, Staging, Development (predefined by backend)
-- **Custom labels**: create explicitly and assign to versions
-- **Name-based APIs**: manage by names (no IDs needed)
-- **Draft safety**: cannot assign labels to drafts; assignments are queued and applied on commit
-
-### Assign labels
-
-
-
-```typescript JS/TS
-// Assign by instance (current project)
-await client.labels().assign("Production", "v1");
-await client.labels().assign("Staging", "v2");
-
-// Create and assign a custom label
-await client.labels().create("Canary");
-await client.labels().assign("Canary", "v2");
-
-// Class helpers by names (org-wide context)
-await Prompt.assignLabelToTemplateVersion("intro-template", "v2", "Development");
-```
-
-```python Python
-# Assign by instance
-client.assign_label("Production", version="v1")
-client.assign_label("Staging", version="v2")
-
-# Create and assign a custom label
-client.create_label("Canary")
-client.assign_label("Canary", version="v2")
-
-# Class helpers by names
-Prompt.assign_label_to_template_version(template_name="intro-template", version="v2", label="Development")
-```
-
-
-
-### Remove labels
-
-
-
-```typescript JS/TS
-await client.labels().remove("Canary", "v2");
-await Prompt.removeLabelFromTemplateVersion("intro-template", "v2", "Development");
-```
-
-```python Python
-client.remove_label("Canary", version="v2")
-Prompt.remove_label_from_template_version(template_name="intro-template", version="v2", label="Development")
-```
-
-
-
-### List labels and mappings
-
-
-
-```typescript JS/TS
-const labels = await client.labels().list(); // system + custom
-const mapping = await Prompt.getTemplateLabels({ template_name: "intro-template" });
-```
-
-```python Python
-labels = client.list_labels()
-mapping = Prompt.get_template_labels(template_name="intro-template")
-```
-
-
-
----
-
-## Fetch by name + label (or version)
-
-
-
-
Precedence: version > label
-
Python default: if no label is provided, defaults to "production"
-
Return type: get_template_by_name() returns a Prompt instance (not a raw PromptTemplate). In Python you can call .compile() directly on it; in TypeScript you wrap the returned template in new Prompt(tpl) then call .compile().
-
-
-
-
-
-```typescript JS/TS
-import { Prompt } from "@future-agi/sdk";
-
-const tplByLabel = await Prompt.getTemplateByName("intro-template", { label: "Production" });
-const tplByVersion = await Prompt.getTemplateByName("intro-template", { version: "v2" });
-```
-
-```python Python
-from fi.prompt import Prompt
-tpl_by_label = Prompt.get_template_by_name("intro-template", label="Production")
-tpl_by_version = Prompt.get_template_by_name("intro-template", version="v2")
-```
-
-
-
----
-
-## A/B testing with labels (compile → OpenAI gpt-4o)
-
-Fetch two labeled versions of the same template (e.g., `prod-a` and `prod-b`), randomly select one, compile variables, and send the compiled messages to OpenAI.
-
-
-The compile() API replaces {`{{var}}`} in string contents and preserves structured contents. Ensure your template contains the variables you pass (e.g., {`{{name}}`}, {`{{city}}`}).
-
-
-
-
-```typescript JS/TS
-import OpenAI from "openai";
-import { Prompt, PromptTemplate } from "@future-agi/sdk";
-
-const openai = new OpenAI({ apiKey: process.env.OPENAI_API_KEY! });
-
-// Fetch both label variants
-const [tplA, tplB] = await Promise.all([
- Prompt.getTemplateByName("my-template-name", { label: "prod-a" }),
- Prompt.getTemplateByName("my-template-name", { label: "prod-b" }),
-]);
-
-// Randomly select a variant
-const selected = Math.random() < 0.5 ? tplA : tplB;
-const client = new Prompt(selected as PromptTemplate);
-
-// Compile variables into the template messages
-const compiled = client.compile({ name: "Ada", city: "Berlin" });
-
-// Send to OpenAI gpt-4o
-const completion = await openai.chat.completions.create({
- model: "gpt-4o",
- messages: compiled as any,
-});
-
-const resultText = completion.choices[0]?.message?.content;
-```
-
-```python Python
-import os
-import random
-
-from openai import OpenAI
-from fi.prompt import Prompt
-
-openai_client = OpenAI(api_key=os.getenv("OPENAI_API_KEY"))
-
-# Fetch both label variants (each returns a Prompt instance)
-client_a = Prompt.get_template_by_name("my-template-name", label="prod-a")
-client_b = Prompt.get_template_by_name("my-template-name", label="prod-b")
-
-# Randomly select a variant
-selected_client = client_a if random.random() < 0.5 else client_b
-
-# Compile variables into the template messages
-compiled = selected_client.compile(name="Ada", city="Berlin")
-
-# Send to OpenAI gpt-4o
-response = openai_client.chat.completions.create(
- model="gpt-4o",
- messages=compiled,
-)
-result_text = response.choices[0].message.content
-# For analytics, log selected_client.template.version or the label (e.g. "prod-a" / "prod-b")
-```
-
-
-
-
-For analytics, attach the selected label/version to your logs or tracing so A/B results can be compared.
-
-
----
-
-## Compile output format
-
-The `compile()` method returns messages in a provider-agnostic format. Each message has `role` and `content`; `content` may be a string or a structured list of parts (e.g. text, images) depending on the SDK and template.
-
-**Example output structure:**
-
-```json
-[
- {"role": "system", "content": "You are a helpful assistant."},
- {"role": "user", "content": "Hello Ada from Berlin!"}
-]
-```
-
-
-If your SDK or backend returns content as a stringified list of content parts (e.g. for multimodal content), you may need an adapter to convert to your target LLM provider’s format (e.g. OpenAI’s role + content string).
-
-
----
-
-## Next Steps
-
-
-
- Build and run prompts in the UI.
-
-
- Generate a prompt draft from a plain-language description.
-
-
- Connect prompts to traces to monitor performance in production.
-
-
- How prompts fit into the platform.
-
-
diff --git a/src/pages/docs/prompt/guides/commit-and-compare-versions.mdx b/src/pages/docs/prompt/guides/commit-and-compare-versions.mdx
new file mode 100644
index 00000000..8ecf76b0
--- /dev/null
+++ b/src/pages/docs/prompt/guides/commit-and-compare-versions.mdx
@@ -0,0 +1,45 @@
+---
+title: "Commit & compare versions"
+description: "Turn a working draft into a version you can point to, then compare a few side by side"
+---
+
+A prompt starts out as a [draft](/docs/prompt/concepts/versions-and-labels) you're still editing. Running it once and committing turns that draft into a version, one that sticks around after you move on to the next edit.
+
+## Commit a version
+
+This picks up with the support-agent template already open in the editor, either fresh from [Create a prompt](/docs/prompt/guides/create-a-prompt) or one you're mid-edit on.
+
+Click **Run Prompt** to produce an output; see [Run a prompt](/docs/prompt/guides/run-a-prompt) for the full walkthrough of messages, models, and variables. Running is also what clears the **Draft** badge next to the version number, which is the same thing that enables **Commit**. Until then, hover the disabled button and the tooltip reads, **"Please run the prompt before saving and committing"**. It also greys out while more than one version is loaded for comparison, and the tooltip still shows that same message.
+
+Click **Commit** in the editor header. This opens the **Commit changes to prompt** dialog. Type a message in the **Commit message** field, something like "Added escalation trigger for refund requests", since both **Commit** and **Commit and set as a default version** stay disabled until you do. Pick **Commit** to save the version as is, or **Commit and set as a default version** to save it and also make it the template's default in the same step.
+
+Once you click **Commit**, the dialog closes and a snackbar confirms it, something like `Commit for successful` (or `...and set as default` if you picked **Commit and set as a default version** instead). The version now shows up under **Commit History**, which lists just the versions you've committed.
+
+## Compare versions
+
+Line up a few versions to see how they differ before you decide which one to promote.
+
+Click **More** in the editor header, then **History**. It lists every version, including drafts you haven't committed yet; **Commit History** narrows that down to the ones you've committed.
+
+Click **Select to compare** to turn on checkboxes next to each version. The version already open in the editor comes pre-checked and locked, and it counts as one of the three, so you're picking at most two more.
+
+Click **Compare**, which appears once you've checked at least one version. Each selected version opens in its own panel, showing its messages, its own last saved output, and its own model configuration side by side, often the detail that differs most between versions.
+
+
+Three is the hard cap. Once three are checked, the remaining checkboxes go disabled with **"Compare limit is upto 3 version only, Deselect other options to select this one"**. Deselect one to swap in another.
+
+
+## Promote a version
+
+Comparing tells you which version should be live. Making it live isn't a commit or compare action, it's a labelling one: you point the Production label at the version you picked instead of changing the version itself. To do it from your own code, use the [SDK & API](/docs/prompt/reference/sdk-api) reference.
+
+## Dive deeper
+
+
+
+ Score outputs across the versions you just compared
+
+
+ Assign labels and fetch versions from your own code
+
+
diff --git a/src/pages/docs/prompt/guides/create-a-prompt.mdx b/src/pages/docs/prompt/guides/create-a-prompt.mdx
new file mode 100644
index 00000000..e003f474
--- /dev/null
+++ b/src/pages/docs/prompt/guides/create-a-prompt.mdx
@@ -0,0 +1,58 @@
+---
+title: "Create a prompt"
+description: "Generate with AI, start from scratch, or use a template, then give the prompt a name"
+---
+
+Every prompt starts in the same place: the [Prompts](/docs/prompt/concepts/understanding-prompts) directory. This guide walks you from there, through the **Create a new prompt** modal, to a named, open editor.
+
+
+
+ In the left navigation, click **Prompts**. You land in the directory, rooted at **All Prompts** and **My templates**.
+
+
+ In the directory toolbar, click **Create prompt**. This opens the **Create a new prompt** modal.
+
+
+ Pick one of the three options in the modal. Each gets its own section below.
+
+
+
+## Pick a starting point
+
+### Generate with AI
+
+Pick this when you don't have the wording yet and want a starting draft to edit. Click **Generate with AI** in the modal: the platform creates the prompt and opens the editor, with the **Generate a prompt** drawer open on top of it. Type a plain-language description of what the prompt should do, for example "write a support agent that answers customer questions using our returns policy," then click **Generate**. Review the generated prompt, then click **Continue** to drop it into the user message. The system message is still yours to write.
+
+### Start from scratch
+
+Pick this when you already know what the [system and user messages](/docs/prompt/concepts/understanding-prompts) should say. Click **Start from scratch** in the modal and the editor opens empty right away: you write the system and user messages yourself.
+
+### Start with a template
+
+Pick this when a team pattern for this kind of prompt already exists. Click **Start with a template** in the modal. The template browser opens instead of the editor: pick a category from the sidebar or search by name, open a template to preview it, then click **Use this template** to load its content into a new prompt.
+
+
+To skip the modal and go straight to the template browser, click **Use template** directly in the directory toolbar instead. It opens the same template browser as this route.
+
+
+## Rename it
+
+The **Generate with AI** and **Start from scratch** routes open the new prompt named `Untitled-1` (or the next free number in your organization). The **Start with a template** route names it `Untitled-1-` instead. Rename it before anything else: click the title in the editor header and type a name, for example `support-agent`.
+
+
+Leaving the name empty shows **Name cannot be empty** and the rename doesn't go through. A name another prompt in your organization already uses is rejected the same way, right when you try the rename.
+
+
+## Dive deeper
+
+
+
+ Write your messages, pick a model, and get a response
+
+
+ Save a version and see what changed
+
+
+ Keep a growing prompt library navigable
+
+
diff --git a/src/pages/docs/prompt/guides/evaluate-prompt-outputs.mdx b/src/pages/docs/prompt/guides/evaluate-prompt-outputs.mdx
new file mode 100644
index 00000000..666015e7
--- /dev/null
+++ b/src/pages/docs/prompt/guides/evaluate-prompt-outputs.mdx
@@ -0,0 +1,62 @@
+---
+title: "Evaluate prompt outputs"
+description: "Read scores alongside outputs, compare them across versions, and remove an eval when you're done with it"
+---
+
+Attach an eval template to a [prompt template](/docs/prompt/concepts/understanding-prompts) and its score shows up right next to the output it scored, inside the same editor you ran the prompt in. This guide continues with support-agent, the same template you ran in [Run a prompt](/docs/prompt/guides/run-a-prompt), and scores the output that run produced. See [Evaluations](/docs/evaluation) for what an eval checks and how it arrives at that score; this page only covers wiring one to a prompt.
+
+## Run the prompt first
+
+The **Evaluation** tab stays disabled until the template has at least one output to score. Open it too early and the tab shows why, in its own tooltip: "You need to submit at least one prompt and get an output before accessing the evaluation." Run support-agent once and the tab unlocks.
+
+## Attach an eval
+
+
+
+ Open the **Evaluation** tab and click **Add Evaluations**.
+
+
+ Choose one from the list, for example `customer_agent_human_escalation`.
+
+ If support-agent doesn't have what the eval needs, pick a different template whose required inputs actually match what support-agent produces.
+
+
+ Point each required input at one of support-agent's own variables (`company_name`, `customer_question`), at `model_input`, or at `model_output`, and give the eval a name.
+
+
+ Leave one required input unmapped and **Save Eval** blocks with "Required input mappings must be filled" until every required input has a target.
+
+
+ Click **Save Eval**. The eval attaches to support-agent and scores the output you already have.
+
+
+
+## Read the results next to each output
+
+Each attached eval's score lands in its own column, next to the output it scored, so you can scan outputs and scores together instead of cross-referencing two views.
+
+The table starts wide: **Show Variables** is on by default, so each variable's value already shows up as its own column, useful when a low score comes from what the model was actually given rather than the model itself. Turn it off to narrow the table down to outputs and scores. **Show Prompts** is off by default; turn it on to add a header band above the output columns showing the prompt messages.
+
+## Run the same evals across several versions
+
+You don't have to reattach an eval to every version by hand. Once the prompt template has more than one [committed version](/docs/prompt/concepts/versions-and-labels), click the **+** in the comparison column header of the results table. In the **Add version to compare** drawer that opens, tick one or two more versions to bring into the table; the current version is already ticked and locked. Click **Compare** and the table now shows outputs and scores side by side for every version you picked.
+
+With several versions in view, reopen **Add Evaluations**: it lists every eval already attached to the template with a checkbox next to each. With nothing checked, the button reads **Run All** and runs every attached eval; check specific ones and it switches to **Run Selected**, running just those. Future AGI runs the evals against every version now in the table, in one pass. Two things can trip this: the template needs at least one eval attached before there's anything to run, and an eval that belongs to a different template than the one you're comparing gets rejected.
+
+## Remove an eval
+
+Open **Add Evaluations** and click the trash icon on the eval's row. Confirm "Delete this evaluation and its results?"; this can't be undone. Removing it drops the eval off the list, and its column and scores disappear from the results table.
+
+## Dive deeper
+
+
+
+ See how a version holds up on real traffic once evals give you a baseline
+
+
+ Generate and score outputs across every row instead of one at a time
+
+
+ Feed the scores you just attached into an algorithm that improves the prompt
+
+
diff --git a/src/pages/docs/prompt/guides/organize-prompts-in-folders.mdx b/src/pages/docs/prompt/guides/organize-prompts-in-folders.mdx
new file mode 100644
index 00000000..a9940d0b
--- /dev/null
+++ b/src/pages/docs/prompt/guides/organize-prompts-in-folders.mdx
@@ -0,0 +1,65 @@
+---
+title: "Organize prompts in folders"
+description: "Create folders, move prompts into them, and rename or delete what you no longer need"
+---
+
+Once a prompt library grows past a handful of templates, a flat list stops working. Group prompts by team, task type, or product area, then use the Prompts directory's own breadcrumbs and sort to get back to one fast.
+
+This guide assumes you already have a prompt to organize, for example `support-agent` from [Create a prompt](/docs/prompt/guides/create-a-prompt); start there if you don't.
+
+
+ New Folder, Move, Rename, and Delete are role-gated. If your role doesn't have permission, **New Folder** appears disabled, and the row menu that holds Move, Rename, and Delete doesn't appear at all.
+
+
+## Create folders and move prompts
+
+
+
+ In the left navigation, click **Prompts** to reach the directory.
+
+
+ In the left tree, click **New Folder**. In the **Create new folder** modal, type a name and click **Create**. You land inside the new folder.
+
+
+ Folder names must be unique within your workspace. If another folder already has that name, the create fails.
+
+
+
+ Click **All Prompts** in the breadcrumbs to get back to the list containing the prompt you want to move.
+
+ Open `support-agent`'s row menu, the three-dot icon at the right end of its row, and select **Move**.
+
+ The modal opens titled **Move "support-agent"**, with a **Select folder** dropdown. The dropdown only moves a prompt from one folder to another. There's no option to unfile a prompt once it's in a folder. Pick the destination and click **Move**. The prompt moves immediately.
+
+
+
+## Rename a folder or a prompt
+
+Open the row menu for a folder or a prompt (`support-agent`, for example) and select **Rename**. Update the name and click **Save**. Both folder names and prompt names must be unique within your workspace, so renaming to a name already in use is rejected.
+
+## Delete a folder or a prompt
+
+Open the row menu and select **Delete**. In the **Delete folder** or **Delete prompt** modal, depending on what you selected, click **Delete**.
+
+
+ Deleting a folder deletes every prompt and [prompt version](/docs/prompt/concepts/versions-and-labels) inside it. Move anything you still need out first.
+
+
+## Browse, sort, and search your library
+
+A few more controls in the directory help you get around once it's grown past a screenful:
+
+- **All Prompts** and **My templates**: the two top-level entries in the left tree. **All Prompts** holds every prompt and folder in your workspace; **My templates** holds prompts you've saved as reusable templates. Click either to jump straight there instead of clicking back through every folder you opened
+- **Sort**: order the current view by **Name** or **Last modified**
+- **Search**: a search bar in the directory, labeled **Search in prompts** on **All Prompts** and **Search in templates** on **My templates**
+
+## Dive deeper
+
+
+
+ Score what the prompt produces
+
+
+ See per-version latency, token, and cost medians
+
+
diff --git a/src/pages/docs/prompt/guides/run-a-prompt.mdx b/src/pages/docs/prompt/guides/run-a-prompt.mdx
new file mode 100644
index 00000000..62572a25
--- /dev/null
+++ b/src/pages/docs/prompt/guides/run-a-prompt.mdx
@@ -0,0 +1,70 @@
+---
+title: "Run a prompt"
+description: "Write messages, set variables and parameters, and run a prompt in the Playground"
+---
+
+The **Playground** tab is where an open template turns into a model response. This guide picks up with a template already open, either a fresh one from [Create a prompt](/docs/prompt/guides/create-a-prompt) or the support-agent prompt built here, using the same running example as [Understanding Prompts](/docs/prompt/concepts/understanding-prompts).
+
+The Playground splits into two panels. The left panel is the **editor**: your messages, with a header row above them for the model picker and its parameters. The toolbar at the top of the workbench holds the **Variables** control and **Run Prompt**. The right panel is the **output panel**, where the response appears once you run the prompt.
+
+## Write the messages
+
+Write the system message and the user message in the editor. For support-agent, the system message sets the agent's role and the user message carries the question it has to answer:
+
+- **System**: "You are a support agent for `{{company_name}}`."
+- **User**: "Answer the following customer question clearly and professionally: `{{customer_question}}`"
+
+`{{company_name}}` and `{{customer_question}}` are variables here. You'll supply values for them in the Variables panel, covered next.
+
+Click **Add Message** to add more, for example an assistant message showing the model a sample answer.
+
+
+The editor always keeps at least one message. Try to remove the only one left and it blocks you with **"You must have at least one prompt."**
+
+
+## Declare variables and supply values
+
+Typing `{{name}}` in any message declares that variable: there's no separate declare step. Open **Variables** in the toolbar, next to **Run Prompt**, to see every variable the template declares.
+
+- Each row holds one full set of values, and each row is one run.
+- **Import Dataset** and **Generate Sample Data** sit above the table if you'd rather not type the rows in by hand.
+
+For support-agent, fill in a row with something like `Acme` for `company_name` and `Do you offer refunds after 30 days?` for `customer_question`.
+
+Every variable needs a value before the prompt runs. If one is still empty, clicking **Run Prompt** opens the Variables panel instead of running.
+
+## Choose a model
+
+Choose the model that runs the prompt, from the model picker in the editor header. There's no default: every prompt needs a model selected before it runs. If none is chosen, clicking **Run Prompt** opens the model picker instead of running.
+
+## Set the parameters
+
+Set the parameters that shape the model's output, things like temperature, max tokens, and top-p, from the same editor header. See [Model configuration](/docs/prompt/reference/model-configuration) for the full list of parameters and their valid ranges.
+
+## Run the prompt and read the output
+
+Click **Run Prompt**. The response fills in the output panel as the model generates it, rather than appearing all at once, so you can start reading before it finishes. For support-agent, a good response is a short, direct answer to `customer_question` that reads as coming from `company_name`, matching the clear, professional tone the messages ask for.
+
+
+If another prompt in your organization already uses the same name, the run stops before it starts. Renaming it clears the conflict.
+
+
+See [Prompt FAQ & fixes](/docs/prompt/troubleshooting) for other reasons a run can fail and how to fix them.
+
+## Stop a run in progress
+
+**Stop Generating** takes the place of **Run Prompt** while a run is in flight, in the same spot in the header. If a run is taking too long or heading somewhere you don't want, click it to cancel.
+
+## Dive deeper
+
+
+
+ The full parameter list with valid ranges
+
+
+ Turn a run you like into a version you can point at
+
+
+ Score what the prompt produces
+
+
diff --git a/src/pages/docs/prompt/guides/track-prompt-performance.mdx b/src/pages/docs/prompt/guides/track-prompt-performance.mdx
new file mode 100644
index 00000000..a2aaa324
--- /dev/null
+++ b/src/pages/docs/prompt/guides/track-prompt-performance.mdx
@@ -0,0 +1,55 @@
+---
+title: "Track prompt performance"
+description: "See how each version of a prompt behaves on real traffic"
+---
+
+The **Metrics** tab in the editor rolls latency, tokens, and cost up per [version](/docs/prompt/concepts/versions-and-labels) for support-agent, so you can compare two versions head to head instead of reading [traces](/docs/observe/concepts/traces) one at a time.
+
+**Before you start:** Metrics only pick up generations that carry a reference to the prompt template they came from. Wire that up first, as covered in [Log prompt templates](/docs/sdk/tracing/log-prompt-templates); this guide picks up once that's in place.
+
+## Run the prompt first
+
+The **Metrics** tab stays disabled until the prompt has produced at least one output. Open it too early and the tab shows why, in its own tooltip: "You need to submit at least one prompt and get an output before accessing the metrics." [Run support-agent once](/docs/prompt/guides/run-a-prompt) and the tab unlocks.
+
+## View per-version metrics
+
+
+
+ In the editor, open support-agent and click the **Metrics** tab. It splits into two sub-tabs: **Metrics**, the per-version table below, and **Linked Traces**, the individual traces behind those numbers.
+
+
+ The table lists one row per version, aggregated across every trace recorded for it:
+
+ | Metric | What it tells you |
+ |---|---|
+ | **Median Latency** | Typical time for the model to produce a response |
+ | **Median Input Tokens** | Typical size of the prompt sent to the model |
+ | **Median Output Tokens** | Typical length of the model's reply |
+ | **Median Cost** | Typical cost per generation for this version |
+ | **No. of traces** | How many times this version was called |
+ | **First Used** | When this version was first called |
+ | **Last Used** | When this version was most recently called |
+ | **Label Name** | Which label, if any, points at this version, useful for telling which one Production is live on |
+
+ If a version you expect doesn't show up, or its trace count looks lower than it should, its generations most likely aren't carrying the template reference. Go back to [Log prompt templates](/docs/sdk/tracing/log-prompt-templates) and check the instrumentation.
+
+
+
+## Drill into the traces behind a number
+
+Switch to the **Linked Traces** sub-tab to see the individual traces that rolled up into those numbers, useful when a median looks off and you want to check what actually produced it.
+
+## Decide if a change helped
+
+Pick a metric, then compare it across two versions. If support-agent v3's median latency comes in lower than v2's, the change helped. If median cost jumps right after you lengthen the system message, that's the number that tells you why.
+
+## Dive deeper
+
+
+
+ Trace production calls as they come in
+
+
+ Assign labels and fetch versions from your own code
+
+
diff --git a/src/pages/docs/prompt/index.mdx b/src/pages/docs/prompt/index.mdx
index e0b15cab..79b4fbc1 100644
--- a/src/pages/docs/prompt/index.mdx
+++ b/src/pages/docs/prompt/index.mdx
@@ -1,67 +1,38 @@
---
-title: "Future AGI Prompt Workbench: Create and Manage AI Prompts"
-description: "Create, manage, version, and optimize AI prompts in the Prompt Workbench for reliable and consistent language model outputs."
+title: "Overview"
+description: "Templates you can version, run against real data, and monitor in production"
---
-## About
+## What is Prompt?
-A prompt is the instruction you give an AI model to produce a response. Getting that instruction right is one of the most impactful things you can do to improve your AI product, but managing prompts without a dedicated tool is messy. They end up hardcoded in application logic, changes are hard to track, and there is no way to compare versions or test them consistently.
+**Prompt** is where your prompt templates live: a named, versioned object on the platform rather than a string hardcoded in your application, so you can change it without shipping a deploy, compare versions side by side, and fetch the current one into your running app with the [SDK](/docs/prompt/reference/sdk-api). The editor is where you compose a template, run it, score it, and watch it in production.
-Prompt Workbench solves this by giving every prompt a permanent, versioned home on the platform. You write prompts using variables so they can accept dynamic inputs at runtime. Every edit creates a new version, and you can compare any two versions side by side or roll back instantly. Prompts are reusable across the entire platform: run them against dataset rows, use them in simulations, include them in experiments, or fetch them from your application via the SDK.
+Open **Prompts** in the left navigation to see your prompt templates. Select or create a [prompt template](/docs/prompt/concepts/understanding-prompts) from there to reach the editor.
-The workbench is also connected to observability. Link a prompt to your production traces and see exactly how it performs on real traffic, closing the loop between what you write and what you ship.
+## How Prompt fits with the rest of the platform
-## How Prompt Connects to Other Features
+From the editor, you can:
-- **Datasets**: Run prompts against dataset rows to generate model outputs at scale. [Learn more](/docs/dataset/features/run-prompt)
-- **Evaluation**: Score prompt outputs with 70+ built-in metrics to measure quality. [Learn more](/docs/evaluation)
-- **Experiments**: Compare prompt versions side by side on the same data. [Learn more](/docs/dataset/features/experiments)
-- **Optimization**: Feed eval scores into optimization algorithms to automatically improve prompts. [Learn more](/docs/optimization)
-- **Observability**: Link prompts to production traces to see latency, cost, and token usage per version. [Learn more](/docs/prompt/features/linked-traces)
+- Run a prompt against [Dataset](/docs/dataset) rows to generate outputs
+- Score its outputs with [Evaluations](/docs/evaluation)
+- Simulate a prompt against scenarios from the editor's [Simulation](/docs/simulation) tab
+- Read its production metrics from [Observe](/docs/observe) traces
-## Getting Started
+## Start here
+
+Create and run a prompt first, then read the concepts behind it.
-
- Build a prompt with full control over structure, variables, and model settings.
-
-
- Start from a pre-built template for common use cases and customize from there.
-
-
- Describe what you need and let the platform generate a prompt to start from.
+
+ From the left navigation to an open, named editor
-
- Connect prompts to production traces to monitor how they perform in the real world.
+
+ Write your messages, pick a model, and get a response
-
- Organize prompts into folders to keep your workspace navigable as it grows.
+
+ The template, message, and version model behind the editor
-
- Fetch and use prompts programmatically from your application via the SDK.
+
+ How a draft becomes a version, and how labels promote it
diff --git a/src/pages/docs/prompt/reference/model-configuration.mdx b/src/pages/docs/prompt/reference/model-configuration.mdx
new file mode 100644
index 00000000..5340361a
--- /dev/null
+++ b/src/pages/docs/prompt/reference/model-configuration.mdx
@@ -0,0 +1,75 @@
+---
+title: "Model configuration"
+description: "Field reference for a prompt template's model configuration and run limits"
+---
+
+Look up a field below for what it accepts and the limit enforced when you save or run it. For which models are available in your workspace, see [AI Providers](/docs/admin-settings/ai-providers) in Admin & Settings.
+
+You set these from the Playground's editor header or the SDK.
+
+## Generation parameters
+
+| Field | Type | Valid range |
+|---|---|---|
+| `temperature` | number | 0.0 to 2.0 |
+| `frequency_penalty` | number | -2.0 to 2.0 |
+| `presence_penalty` | number | -2.0 to 2.0 |
+| `top_p` | number | 0.0 to 1.0 |
+| `max_tokens` | integer | 1 to 65536 |
+
+A value outside its range is rejected before the model runs. These ranges are what the API and SDK enforce.
+
+## Output and tool settings
+
+| Field | Type | Valid values | Default |
+|---|---|---|---|
+| `output_format` | string | `array`, `string`, `number`, `object`, `audio`, `image` | `string` |
+| `tool_choice` | string | `auto`, `required`, or unset | Unset |
+| `tools` | array | A list of tool IDs | Unset |
+| `response_format` | object | A JSON schema object, a string, or the ID of a saved [response schema](#response-schemas) | Unset |
+
+
+ An ID that doesn't resolve to a real tool or response schema is rejected, not silently dropped.
+
+
+## Prompt structure limits
+
+| Item | Limit |
+|---|---|
+| Messages | An ordered list of `{role, content}` objects. `role` must be `system`, `user`, or `assistant`; `content` must be a string. Any other role is rejected |
+| Prompt template name | Up to 2000 characters |
+| Base template name | Up to 255 characters |
+| Folder name | Up to 255 characters |
+| Version name | Must match `v` followed by digits, for example `v1`, `v12`. Anything else is rejected |
+| Version comparison | Up to 3 versions at once. A fourth is rejected |
+
+## Run behavior limits
+
+Limits on how many calls run at once from a [Run Prompt column](/docs/dataset/guides/run-a-prompt-on-every-row) across a dataset.
+
+| Item | Limit |
+|---|---|
+| Concurrent calls in a dataset run | 1 to 10, default 5. A value over 10 is rejected |
+
+## Response schemas
+
+A response schema is a saved shape that `response_format` can point at by ID instead of an inline JSON schema object.
+
+| Field | What it means |
+|---|---|
+| `name` | Unique within your organization and workspace |
+| `schema_type` | `json` or `yaml` |
+
+## Keep exploring
+
+
+
+ Score and compare what a prompt returns
+
+
+ Set these parameters in the Playground
+
+
+ Set the same fields from code
+
+
diff --git a/src/pages/docs/prompt/reference/sdk-api.mdx b/src/pages/docs/prompt/reference/sdk-api.mdx
new file mode 100644
index 00000000..38433f67
--- /dev/null
+++ b/src/pages/docs/prompt/reference/sdk-api.mdx
@@ -0,0 +1,265 @@
+---
+title: "SDK & API"
+description: "Prompt SDK calls for templates, versions, labels, and compile"
+---
+
+## Prompts from code
+
+From code, you can build a [prompt template](/docs/prompt/concepts/understanding-prompts), move it through drafts and versions, point [labels](/docs/prompt/concepts/versions-and-labels) at a version, fetch it by name, and compile it into messages ready for a model call. Execution support was removed from both SDKs: neither exposes a run method, so running a prompt or comparing versions stays in the editor. This page is the call reference for everything else.
+
+## Install and authenticate
+
+
+
+```bash Python
+pip install futureagi
+```
+
+```bash TypeScript
+npm install @future-agi/sdk
+```
+
+
+
+
+The Python package is published as **`futureagi`** but imported as **`fi`**, for example `from fi.prompt import Prompt`.
+
+
+Every call on this page needs your Future AGI credentials:
+
+```bash
+export FI_API_KEY="your-api-key"
+export FI_SECRET_KEY="your-secret-key"
+```
+
+Find both under **Settings → API Keys** in the platform.
+
+## Construct a template
+
+| Field | Holds |
+|---|---|
+| `name` | The template's name |
+| `messages` | Ordered `SystemMessage` / `UserMessage` / `AssistantMessage` objects, plus any placeholder entries |
+| `model_configuration` | A `ModelConfig`: model name and generation settings. See [Model configuration](/docs/prompt/reference/model-configuration) for every field and its valid range |
+| `variable_names` | Sample values for each `{{name}}` the messages reference |
+| `placeholders` | Named slots that take a list of messages instead of a string, set as a `{"type": "placeholder", "name": ...}` entry in `messages` |
+
+
+
+```python Python
+from fi.prompt import Prompt, PromptTemplate, ModelConfig, SystemMessage, UserMessage
+
+template = PromptTemplate(
+ name="support-agent",
+ messages=[
+ SystemMessage(content="You are a support agent for {{company_name}}"),
+ {"type": "placeholder", "name": "history"},
+ UserMessage(content="{{customer_question}}"),
+ ],
+ model_configuration=ModelConfig(model_name="gpt-4o-mini"),
+ variable_names={
+ "company_name": ["Acme"],
+ "customer_question": ["Where is my order?"],
+ },
+)
+```
+
+```typescript TypeScript
+import { Prompt, PromptTemplate, ModelConfig, SystemMessage, UserMessage } from "@future-agi/sdk";
+
+const template = new PromptTemplate({
+ name: "support-agent",
+ messages: [
+ new SystemMessage("You are a support agent for {{company_name}}"),
+ { type: "placeholder", name: "history" } as any, // PromptTemplate.messages is typed MessageBase[], so a placeholder entry needs the cast
+ new UserMessage("{{customer_question}}"),
+ ],
+ model_configuration: new ModelConfig({ model_name: "gpt-4o-mini" }),
+ variable_names: {
+ company_name: ["Acme"],
+ customer_question: ["Where is my order?"],
+ },
+});
+```
+
+
+
+## Version lifecycle
+
+A template moves through the same states from code as it does in the editor: draft, commit, new draft.
+
+| Step | Python | TypeScript |
+|---|---|---|
+| Create or open a draft | `Prompt(template=template).create()` | `await new Prompt(template).open()` |
+| Save changes to the current draft | `client.save_current_draft()` | `await client.saveCurrentDraft()` |
+| Commit, optionally set default and a label | `client.commit_current_version(message="...", set_default=True, label="Production")` | `await client.commitCurrentVersion("...", true, "Production")` |
+| Open a new draft version | `client.create_new_version(commit_message="...", set_default=True)` | `await client.createNewVersion({ commit_message: "...", set_default: true })` |
+| Set an already-committed version as default | `Prompt.set_default_version(template_name="support-agent", version="v2")` | `await Prompt.setDefaultVersion("support-agent", "v2")` |
+| Delete the template | `client.delete()` | `await client.delete()` |
+| Delete the template by name | `Prompt.delete_template_by_name("support-agent")` | `await Prompt.deleteTemplateByName("support-agent")` |
+
+Wherever a call takes a `version`, pass it as `v` followed by the version number, for example `"v1"` or `"v2"`. Any other shape is rejected.
+
+
+In Python, constructing `Prompt(template=template)` looks the template up by name first. In TypeScript, `new Prompt(template)` does no lookup: it only assigns the template, and the name lookup happens inside `open()`, which adopts the existing template and returns the client rather than raising. In Python, calling `create()` on a name that already exists raises `TemplateAlreadyExists`; use `get_template_by_name()` to open an existing one instead.
+
+
+
+
+```python Python
+client = Prompt(template=template)
+client.create() # draft v1
+client.save_current_draft() # push further edits to the v1 draft
+client.commit_current_version(
+ message="Add escalation instructions",
+ set_default=True,
+ label="Production",
+)
+client.create_new_version(
+ commit_message="Tune temperature",
+ set_default=False,
+) # commits v1 if still a draft, then opens v2
+```
+
+```typescript TypeScript
+const client = new Prompt(template);
+await client.open(); // draft v1
+await client.saveCurrentDraft(); // push further edits to the v1 draft
+await client.commitCurrentVersion("Add escalation instructions", true, "Production");
+await client.createNewVersion({
+ commit_message: "Tune temperature",
+ set_default: false,
+}); // commits v1 if still a draft, then opens v2
+```
+
+
+
+
+`save_current_draft()` / `saveCurrentDraft()` only work on a draft. Called against a version that's already committed, both raise: create a new draft version first.
+
+
+## Labels
+
+Three system labels are available to every template: **Production**, **Staging**, and **Development**. A custom label works the same way once you create it.
+
+
+
+```python Python
+client.create_label("Canary")
+client.assign_label("Canary", version="v2")
+client.remove_label("Canary", version="v2")
+labels = client.list_labels()
+```
+
+```typescript TypeScript
+await client.labels().create("Canary");
+await client.labels().assign("Canary", "v2");
+await client.labels().remove("Canary", "v2");
+const labels = await client.labels().list();
+```
+
+
+
+
+Assigning a label to the version the client currently has open doesn't fail even while that version is still a draft: `assign_label()` / `labels().assign()` queue the assignment and apply it automatically on your next commit. Pass any other version and the assignment applies immediately instead of queueing.
+
+
+The name-based class helpers skip loading a template instance first; they resolve everything by name, including the version:
+
+
+
+```python Python
+Prompt.assign_label_to_template_version(template_name="support-agent", version="v2", label="Development")
+Prompt.remove_label_from_template_version(template_name="support-agent", version="v2", label="Development")
+Prompt.get_template_labels(template_name="support-agent")
+```
+
+```typescript TypeScript
+await Prompt.assignLabelToTemplateVersion("support-agent", "v2", "Development");
+await Prompt.removeLabelFromTemplateVersion("support-agent", "v2", "Development");
+await Prompt.getTemplateLabels({ template_name: "support-agent" });
+```
+
+
+
+
+Only `assign_label_to_template_version()` / `assignLabelToTemplateVersion()` checks this: pointing it at a version that's still a draft raises an error telling you to commit first, instead of queueing like `assign_label()` does. `remove_label_from_template_version()` / `removeLabelFromTemplateVersion()` and `get_template_labels()` / `getTemplateLabels()` don't check at all.
+
+
+## Fetch by name
+
+An explicit `version` wins over an explicit `label`. Pass neither, and both SDKs fall back to whatever the **Production** label currently points at; if nothing carries that label yet, they fall back again to the template's default version.
+
+
+**Return type differs.** Python's `get_template_by_name()` returns a `Prompt` instance, so you can call `.compile()` on it directly. TypeScript's `getTemplateByName()` returns a raw `PromptTemplate`; wrap it in `new Prompt(tpl)` before calling `.compile()`.
+
+
+
+
+```python Python
+by_version = Prompt.get_template_by_name("support-agent", version="v2")
+by_label = Prompt.get_template_by_name("support-agent", label="Staging")
+by_default = Prompt.get_template_by_name("support-agent") # Production, then the default version
+```
+
+```typescript TypeScript
+const byVersion = await Prompt.getTemplateByName("support-agent", { version: "v2" });
+const byLabelTpl = await Prompt.getTemplateByName("support-agent", { label: "Staging" });
+const byLabel = new Prompt(byLabelTpl); // wrap: getTemplateByName returns a PromptTemplate, not a Prompt
+const byDefaultTpl = await Prompt.getTemplateByName("support-agent"); // Production, then the default version
+const byDefault = new Prompt(byDefaultTpl);
+```
+
+
+
+## Compile
+
+`compile()` substitutes each `{{name}}` in the message content with the value you pass, and expands a placeholder entry into the list of messages you supply for it.
+
+
+
+```python Python
+compiled = client.compile(
+ company_name="Acme",
+ customer_question="Where is my order?",
+ history=[{"role": "user", "content": "I ordered a jacket yesterday."}],
+)
+```
+
+```typescript TypeScript
+const compiled = client.compile({
+ company_name: "Acme",
+ customer_question: "Where is my order?",
+ history: [{ role: "user", content: "I ordered a jacket yesterday." }],
+} as any);
+```
+
+
+
+Both return a flat list of `{role, content}` messages, the placeholder's messages inlined at the position it occupied in the template:
+
+```json
+[
+ { "role": "system", "content": "You are a support agent for Acme" },
+ { "role": "user", "content": "I ordered a jacket yesterday." },
+ { "role": "user", "content": "Where is my order?" }
+]
+```
+
+
+A history item missing `role` or `content` raises a `ValueError` in Python naming the placeholder. And the two SDKs shape `content` differently: Python's `compile()` always returns a string, so structured or multimodal content gets stringified rather than preserved; TypeScript's `compile()` keeps structured content as a list of parts and substitutes only the text fields.
+
+
+## Keep exploring
+
+
+
+ Walk the version lifecycle from the editor
+
+
+ Errors like `TemplateAlreadyExists` and other call failures, explained
+
+
+ Every field on `model_configuration` and its valid range
+
+
diff --git a/src/pages/docs/prompt/troubleshooting.mdx b/src/pages/docs/prompt/troubleshooting.mdx
new file mode 100644
index 00000000..7362474a
--- /dev/null
+++ b/src/pages/docs/prompt/troubleshooting.mdx
@@ -0,0 +1,48 @@
+---
+title: "Prompt FAQ & fixes"
+description: "Symptoms, causes, and fixes for the errors you hit in Prompt"
+---
+
+## In this page
+
+The blocked states and errors you actually run into in Prompt, with the cause and the fix for each. Find your symptom in the tables below: [Blocked buttons and tabs](#blocked-buttons-and-tabs), [Error messages](#error-messages), or [Permissions, plan, and credits](#permissions-plan-and-credits). Output that's wrong, inconsistent, or off-tone is a prompt-engineering problem, not a bug: see [Prompt Engineering](/docs/prompt/concepts/prompt-engineering). Not seeing your symptom? [Contact us](https://futureagi.com/contact-us) and we'll help you track it down.
+
+## Blocked buttons and tabs
+
+| Symptom | Cause | Fix |
+|---|---|---|
+| Clicking the **Evaluation**, **Metrics**, or **Simulation** tab does nothing | The tab needs a run version with content and no unsaved edit: there's no version yet, the selected version is an unsaved draft, the prompt content is empty, or a run is still generating | Run the prompt to completion in the [Playground](/docs/prompt/guides/run-a-prompt) without further edits, then open the tab; editing again re-locks it |
+| **Commit** is disabled | The prompt hasn't been run yet, more than one version is selected, or you're on the **Evaluation** or **Metrics** tab; that last case has no tooltip explaining it | Run the prompt if it hasn't been run, select a single version if more than one is checked, or switch back to the **Playground** tab, then open [Commit](/docs/prompt/guides/commit-and-compare-versions) |
+| Checking a fourth version to compare is disabled, with a tooltip reading "Compare limit is upto 3 version only, Deselect other options to select this one" | Three versions are already selected; comparison is capped at three | Deselect one of the checked versions, then check the one you want to [compare](/docs/prompt/guides/commit-and-compare-versions#compare-versions) |
+
+## Error messages
+
+| Symptom | Cause | Fix |
+|---|---|---|
+| You press **Run** and get "A template with this name already exists." (renaming into a taken name triggers the same check) | Another template in your organization already has that name | Pick a name that's unique in your organization |
+| "Name cannot be empty" | You tried to save a rename with the name field blank | Type a name before saving |
+| "You must have at least one prompt." | You tried to remove the only message left in the editor | Add a replacement message before deleting the last one |
+| "Audio input is missing. Please add audio before running the prompt." | You ran an audio-capable model without attaching audio content | Attach audio to the prompt, then run it |
+
+## Permissions, plan, and credits
+
+| Symptom | Cause | Fix |
+|---|---|---|
+| **Create prompt** and **Use template** appear greyed out in the Prompts directory | Your role is Viewer, which is read-only on Prompts | Ask an Owner or Admin to move you to Member or above; see [Roles & Permissions](/docs/roles-and-permissions) |
+| The three-dot menu (Move, Rename, Delete) doesn't appear on a prompt | Either your role is Viewer, which is read-only on Prompts, or the prompt is a sample prompt, which hides the menu for every role | Ask an Owner or Admin to move you to Member or above if it's a role issue; a sample prompt can't be moved, renamed, or deleted by anyone |
+| **Generate with AI** (or another AI-assisted action) fails | Your organization is out of credits | Top up your wallet; see [Billing & Pricing](/docs/admin-settings/billing-pricing) |
+| **Improve prompt** isn't available | It ships behind a licensed feature | Contact [Future AGI](https://futureagi.com/contact-us) to enable it |
+
+## Keep exploring
+
+
+
+ Fix output quality, not product errors
+
+
+ Write messages, set variables, and get a response
+
+
+ Turn a draft into a version you can point at
+
+
diff --git a/src/pages/docs/protect/concepts/concept.mdx b/src/pages/docs/protect/concepts/concept.mdx
deleted file mode 100644
index df321c64..00000000
--- a/src/pages/docs/protect/concepts/concept.mdx
+++ /dev/null
@@ -1,88 +0,0 @@
----
-title: "Future AGI Protect Use Cases: Security and Guardrails"
-description: "Future AGI Protect safeguards AI applications with real-time guardrails for security, reliability, and compliance across text, image, and audio modalities."
----
-
-By combining custom screening logic with Future AGI's specialized safety models, Protect enables teams to instantly detect, flag, and mitigate risks across four safety dimensions, enhancing the integrity of AI applications without compromising performance.
-
-## **Key Use Cases**
-
-Protect operates across four essential safety dimensions: **Content Moderation** (toxicity and harmful language), **Bias Detection** (sexism and discrimination), **Security** (prompt injection and adversarial attacks), and **Data Privacy Compliance** (PII detection and regulatory adherence). These categories work together to provide comprehensive protection for enterprise AI deployments.
-
-### **1. Content Moderation on Social Media Platforms**
-
-Social media platforms process millions of user interactions daily, making moderation a major challenge. Protect helps by:
-
-- Flagging harmful or inappropriate content in real time across text, images, and videos
-- Detecting hate speech, misinformation, and abusive language
-- Preventing the spread of illegal or unethical materials
-- Preserving genuine engagement while maintaining safe interactions
-
-### **2. Securing AI-Powered Customer Support**
-
-AI chatbots and virtual assistants are often the first point of contact for users. Protect enhances their safety by:
-
-- Blocking spam, phishing attempts, and malicious queries
-- Identifying abusive or harmful user inputs to protect agents
-- Defending against prompt injection attacks that could manipulate AI behavior
-- Screening text and voice-based messages in real time for policy violations across chat and voice agents
-
-### **3. Enforcing Safety & Compliance in Healthcare AI**
-
-Healthcare AI must meet strict regulatory and ethical standards. Protect supports this by:
-
-- Filtering unverified medical advice and health misinformation
-- Preventing AI systems from delivering harmful or misleading responses
-- Protecting sensitive patient data from exposure
-- Enabling compliance with HIPAA and other global healthcare regulations
-
-### **4. Preventing Bias and Ethical Violations**
-
-Fairness is essential in AI-powered decision-making. Protect helps uphold ethical standards by:
-
-- Detecting bias in outputs related to hiring, lending, or other critical decisions
-- Promoting fairness and transparency in AI recommendations
-- Identifying and mitigating harmful stereotypes in generated content
-
-### **5. Real-Time Threat Detection in Cybersecurity**
-
-AI systems in security-critical environments must act fast. Protect strengthens defences by:
-
-- Detecting prompt injection and adversarial manipulation
-- Screening for suspicious or abnormal user behavior
-- Safeguarding models against malicious inputs and misuse
-
-### **6. Protecting Children in Educational AI**
-
-Educational AI tools must be built with child safety in mind. Protect ensures:
-
-- Inappropriate or unsafe content is filtered in real time
-- Compliance with COPPA and other child protection laws
-- Learning environments remain safe, ethical, and age-appropriate
-
-### **7. Ensuring Safety in Voice-Activated Systems**
-
-Voice-enabled AI applications like virtual assistants, smart devices, and IVR systems require real-time monitoring to prevent misuse. Protect enhances safety in audio-first experiences by:
-
-- Detecting inappropriate, harmful, or unsafe voice inputs and outputs
-- Screening spoken content for policy violations or abuse
-- Enabling safer, more reliable voice interactions in homes, cars, and public environments
-
-### 8. Visual Content Safety for Image-Based Applications
-
-Applications that process user-generated images—from social media to content management systems—need robust visual content moderation. Protect provides:
-
-- Real-time detection of inappropriate, violent, or harmful visual content
-- Screening for bias and discrimination in images and memes
-- Privacy protection by identifying and flagging images containing sensitive information
-- Comprehensive safety for platforms handling visual user-generated content
-
-### **Conclusion**
-
-As AI applications become more deeply integrated into everyday life, the need for robust, real-time safeguards grows exponentially. Future AGI's Protect is more than a guardrail—it's a foundational layer that reinforces the security, reliability, and ethical integrity of AI systems in production.
-
-By acting as a live filter across text, image, and audio interactions, Protect enables teams to detect and mitigate risks instantly—whether moderating harmful language in chat, screening visual content for violations, blocking unsafe audio prompts in voice assistants, or ensuring regulatory compliance across all channels.
-
-Built on Google's efficient Gemma 3n architecture with specialized fine-tuned adapters for each safety dimension, Protect delivers state-of-the-art accuracy while maintaining the low latency required for production environments. With native multi-modal support, Protect empowers teams to deploy AI applications that are safe, compliant by default, and trusted by design. As AI continues to evolve, Protect remains your vital safeguard for responsible and future-ready AI deployment.
-
----
\ No newline at end of file
diff --git a/src/pages/docs/protect/concepts/understanding-protect.mdx b/src/pages/docs/protect/concepts/understanding-protect.mdx
new file mode 100644
index 00000000..76afd079
--- /dev/null
+++ b/src/pages/docs/protect/concepts/understanding-protect.mdx
@@ -0,0 +1,69 @@
+---
+title: "Understanding Protect"
+description: "Protect screens your traffic for safety issues through guardrails and protect()."
+---
+
+## A guardrail is a named check on your traffic
+
+A **guardrail** wraps a check and adds three settings you configure in [Agent Command Center](/docs/command-center/features/guardrails), under **Gateway** > **Guardrails**:
+
+- An **action** for when it triggers: Block, Warn, Mask, or Log
+- A **stage** for when it runs: **pre** runs before the model sees the request, **post** runs before the response reaches the caller, and **both** runs on each side
+- A **confidence threshold** for how sure the check has to be before it fires
+
+A stage of **both** isn't a third mode. It's the same check running twice, once on the way in and once on the way out. Stage belongs to the guardrail rather than to the check it wraps, which is why a single **gateway**, the Agent Command Center surface your guardrails live on, can carry a mix: some watching only what the caller sent, others only what the model sent back.
+
+The threshold runs from 0.0 to 1.0: raising it makes the check fire only on higher-confidence matches, lowering it makes it fire more readily. Threshold and action are independent settings. The threshold decides how often a check fires; the action decides what the caller experiences when it does. Loosening a threshold on a check set to Log changes nothing a caller can see. For the response status each action produces, see [Guardrail checks](/docs/protect/reference/guardrail-checks#response-statuses).
+
+Every guardrail wraps a check, and the Rules tab groups checks under two headers that mark a single split: where the check runs. **Rule-Based Checks** are [first-party](/docs/protect/reference/guardrail-checks): they run inside Protect without an external provider. **AI-Powered Checks** call a provider you configure under **Provider Settings** in that check's own settings.
+
+ A["Action: Block, Warn, Mask, or Log"]
+ G --> S["Stage: pre, post, or both"]
+ G --> T["Confidence threshold"]
+ G --> C["Check"]
+ C --> RBC["Rule-Based: runs inside Protect"]
+ C --> AIP["AI-Powered: external provider"]`} />
+
+## protect() screens one input in your code
+
+Separate from the dashboard sits a second surface: [`protect()`](/docs/protect/guides/run-protect-from-the-sdk), a function you call directly in your own code. It screens one input at a time, text, image, or audio, against a fixed, shorter list of checks: toxicity, bias, prompt injection, and data privacy (PII).
+
+`protect()` does not read your dashboard's guardrail configuration. Disabling or customizing a check on the Rules tab has no effect on what `protect()` screens for, and calling `protect()` doesn't touch anything on the dashboard either. The two surfaces are configured independently.
+
+### When to use each
+
+- **Dashboard guardrails**: blanket coverage of every request in your org's traffic, configured once in Agent Command Center
+- **`protect()`**: screening one specific input inline in your own code
+
+## A worked example: PII Detection through a guardrail
+
+A support agent's traffic has a guardrail named PII Detection enabled, set to Block. A request from a customer carries an email address. Its confidence clears the threshold, the check fires, and the Block action stops the request before it reaches the model.
+
+### Where the verdict ends up
+
+That outcome doesn't disappear once the request is blocked. Every request is recorded with whether a guardrail triggered and the result of each check that ran, and that record is what shows up in Logs and Analytics for this request. From there, feedback on the verdict can mark it correct or wrong, telling you whether PII Detection's threshold and action are actually tuned right, not just switched on.
+
+This record belongs to the dashboard-guardrail surface, while `protect()` returns its verdict directly to your code as the function's response; see the [Protect SDK reference](/docs/sdk/protect) for the response shape.
+
+## Why it matters
+
+Treating a guardrail and `protect()` as one surface means assuming a check you configured in one place is protecting you in the other, when it isn't.
+
+Two guardrails with the same check can still behave completely differently once their action, stage, or threshold diverge, because a guardrail is this fixed shape, not a single on/off switch.
+
+## Keep exploring
+
+
+
+ Enable a check, set its stage, threshold, and action
+
+
+ Browse the full list of first-party and provider-backed checks
+
+
+ Full reference for the protect() module
+
+
diff --git a/src/pages/docs/protect/features/run-protect.mdx b/src/pages/docs/protect/features/run-protect.mdx
deleted file mode 100644
index 093b7f7c..00000000
--- a/src/pages/docs/protect/features/run-protect.mdx
+++ /dev/null
@@ -1,177 +0,0 @@
----
-title: "Run Future AGI Protect via SDK for Real-Time Safety Checks"
-description: "Set up and configure Future AGI Protect to apply real-time safety checks to your AI application's inputs and outputs using the SDK."
----
-
-## About
-
-**Run Protect via SDK** is the programmatic interface to Future AGI's real-time guardrailing system. It exposes rule-based safety checks across four dimensions: Content Moderation, Bias Detection, Security, and Data Privacy Compliance: for text, image, and audio inputs, with configurable actions, explanations, and fail-fast evaluation behavior.
-
----
-
-## When to use
-
-- **Block toxic or harmful content**: Screen user inputs or model outputs for hate speech, threats, and harassment before they reach end users.
-- **Detect prompt injection**: Catch adversarial attempts to override instructions or manipulate your AI system.
-- **Enforce data privacy**: Automatically flag PII (names, emails, phone numbers, SSNs) to stay GDPR/HIPAA compliant.
-- **Multi-modal safety**: Apply the same rules to text, image URLs, and audio files without separate pipelines.
-- **Apply multiple rules at once**: Bundle all four safety dimensions in one call for comprehensive protection.
-
----
-
-## How to
-
-
-
- Set up your Future AGI account and get started with Future AGI's robust SDKs. Follow the QuickStart guide:
-
-
- Click [here](https://docs.futureagi.com/admin-settings#accessing-api-keys) to learn how to access your API key.
-
-
-
- To begin using Protect initialize the Protect instance. This will handle the communication with the API and apply defined safety checks.
-
- ```python
- from fi.evals import Protect
-
- # Initialize Protect client (uses environment variables FI_API_KEY and FI_SECRET_KEY)
- protector = Protect()
-
- # Or initialize with explicit credentials
- protector = Protect(
- fi_api_key="your_api_key_here",
- fi_secret_key="your_secret_key_here"
- )
- ```
-
- Protect automatically reads `FI_API_KEY` and `FI_SECRET_KEY` from your environment variables if not explicitly provided.
-
-
- The `protect()` method accepts several arguments and rules to configure your protection checks.
-
- **Arguments:**
-
- | Argument | Type | Default Value | Description |
- | --- | --- | --- | --- |
- | `inputs` | `string` or `list[string]` |: | Input to be evaluated. Can be text, image URL/path, audio URL/path, or data URI |
- | `protect_rules` | `List[Dict]` |: | List of safety rules to apply |
- | `action` | `string` | `"Response cannot be generated as the input fails the checks"` | Custom message shown when a rule fails |
- | `reason` | `bool` | `False` | Include detailed explanation of why content failed |
- | `timeout` | `int` | `30000` | Max time in milliseconds for evaluation |
-
- Rules are defined as a list of dictionaries. Each rule specifies which safety dimension to check.
-
- | Key | Required | Type | Values | Description |
- | --- | --- | --- | --- | --- |
- | `metric` | yes | `string` | `content_moderation`, `bias_detection`, `security`, `data_privacy_compliance` | Which safety dimension to check |
- | `action` | no | `string` | Any custom message | Override the default action message for this specific rule |
-
- ```python
- rules = [
- {"metric": "content_moderation"},
- {"metric": "bias_detection"},
- {"metric": "security"},
- {"metric": "data_privacy_compliance"}
- ]
- ```
-
- - Evaluation stops as soon as **one rule fails** (fail-fast behavior)
- - Rules are processed in parallel batches for optimal performance
- - All four safety dimensions work across text, image, and audio modalities
-
-
- Call `protector.protect()` with your input and rules. When a check is run, a response dictionary is returned with detailed results.
-
- | Key | Type | Description |
- | --- | --- | --- |
- | `status` | `string` | `"passed"` or `"failed"` - result of rule evaluation |
- | `messages` | `string` | Custom action message (if failed) or original input (if passed) |
- | `completed_rules` | `list[string]` | Rules that were successfully evaluated |
- | `uncompleted_rules` | `list[string]` | Rules skipped due to early failure or timeout |
- | `failed_rule` | `list[string]` | Which rule(s) caused the failure (empty if passed) |
- | `reasons` | `list[string]` | Explanation(s) of failure or `["All checks passed"]` |
- | `time_taken` | `float` | Time taken in seconds |
-
- ```python
- result = protector.protect(
- "AI Generated Message",
- protect_rules=rules,
- action="I'm sorry, I can't help you with that.",
- reason=True,
- timeout=25000
- )
- print(result)
- ```
-
- **Pass:**
- ```python
- {
- 'status': 'passed',
- 'completed_rules': ['content_moderation', 'bias_detection'],
- 'uncompleted_rules': [],
- 'failed_rule': [],
- 'messages': 'I like apples',
- 'reasons': ['All checks passed'],
- 'time_taken': 0.234
- }
- ```
-
- **Fail:**
- ```python
- {
- 'status': 'failed',
- 'completed_rules': ['content_moderation', 'bias_detection'],
- 'uncompleted_rules': ['security', 'data_privacy_compliance'],
- 'failed_rule': ['data_privacy_compliance'],
- 'messages': 'Response cannot be generated as the input fails the checks',
- 'reasons': ['Content contains personally identifiable information'],
- 'time_taken': 0.156
- }
- ```
-
-
- Protect natively supports text, image, and audio inputs. Pass your input as a string: the system auto-detects the type.
-
- ```python
- # Image URL
- result = protector.protect(
- "https://example.com/image-sample",
- protect_rules=[{"metric": "content_moderation"}, {"metric": "bias_detection"}],
- action="Image cannot be displayed",
- reason=True,
- timeout=25000
- )
-
- # Audio file path
- result = protector.protect(
- "/path/to/local/audio.wav",
- protect_rules=[{"metric": "content_moderation"}, {"metric": "bias_detection"}],
- action="Audio content cannot be processed",
- reason=True,
- timeout=25000
- )
- ```
-
- **Supported formats:**
- - Images: JPG, PNG, WebP, GIF, BMP, TIFF, SVG: URL, file path, or data URI
- - Audio: MP3, WAV: URL, file path, or data URI
- - Local files are auto-converted to data URIs (max 20 MB). Use direct download URLs, not preview links.
-
-
-
----
-
-## Next Steps
-
-
-
- Full SDK reference for the Protect module.
-
-
- Real-world use cases across content moderation, healthcare, security, and more.
-
-
- Apply safety checks as gateway guardrails for all LLM traffic.
-
-
diff --git a/src/pages/docs/protect/guides/review-guardrail-activity.mdx b/src/pages/docs/protect/guides/review-guardrail-activity.mdx
new file mode 100644
index 00000000..6c1666a3
--- /dev/null
+++ b/src/pages/docs/protect/guides/review-guardrail-activity.mdx
@@ -0,0 +1,52 @@
+---
+title: "Review guardrail activity"
+description: "Read a guardrail's trigger volume and latency, then correct a verdict it got wrong from the request log."
+---
+
+Guardrails fire on every request that matches their rules, and four surfaces tell you how that's going: the **Analytics** tab for volume and latency, the **Logs** tab for the individual requests that triggered a check, the request drawer for correcting a verdict a guardrail got wrong, and the **Feedback** tab for accuracy over time. This guide walks through all four.
+
+This assumes Agent Command Center, where Protect's guardrails live under Gateway, is already set up with at least one provider connected. If it isn't yet, start with the [Agent Command Center Quickstart](/docs/command-center/quickstart). It also assumes you already have a guardrail turned on and that it's fired on at least one request. If not, [turn on a guardrail](/docs/protect/guides/turn-on-a-guardrail) first, then come back once it's had a chance to fire. Until it does, the trends chart, rules table, Logs tab, and Feedback tab covered below all show empty states instead of data.
+
+## Check trigger volume in Analytics
+
+Go to **Gateway** > **Guardrails** > **Analytics**. A range toggle switches the whole tab between **24h**, **7d**, and **30d**, so start by picking the window you care about.
+
+Four numbers sit at the top. Together they say how often guardrails are firing, what they did when they fired, and what that cost in latency:
+
+| Metric | What it measures |
+|---|---|
+| Trigger Rate | Percentage of requests in the range where any guardrail triggered |
+| Blocked | How many checks blocked, not requests (a request with two blocking checks counts twice) |
+| Warned | How many checks warned, not requests (a request with two warning checks counts twice) |
+| Avg Latency | Average latency added by the guardrail check itself, not the total request latency |
+
+Below the numbers, the **Guardrail Triggers Over Time** chart plots trigger volume across the selected range, so you can spot a spike or a change in behavior.
+
+The **Top Triggered Rules** table breaks that volume down by rule. Use it to see which check is driving the volume, and whether that check is mostly blocking or mostly warning. One rule taking up most of the Share is usually the one worth investigating first, and that's what the next section walks through.
+
+## Find the requests in Logs
+
+Once you know which check is generating volume, go to **Gateway** > **Guardrails** > **Logs** to see the actual requests it fired on. This isn't scoped to the check you picked in Top Triggered Rules; it lists recent guardrail-triggered requests across the gateway.
+
+Click a row to jump to that request in the Gateway request logs, the dashboard's full log of request traffic rather than a guardrail-specific view. Click the request there to open its detail drawer.
+
+## Correct a verdict in the request drawer
+
+The request's detail drawer is where you tell Future AGI whether the guardrail got it right. The feedback controls sit on the drawer's **Guardrails** tab, with one set per check that fired, so a verdict attaches to a specific check rather than to the request as a whole. Mark the verdict as one of **Correct**, **False Positive**, **False Negative**, or **Unsure**. Add a comment if you want to note why, then press **Submit Feedback**. Edit is available right after you submit.
+
+Feedback is recorded and summarized. It does not change how the check behaves: submitting a False Positive doesn't loosen the rule or stop it from firing on similar requests going forward. To actually stop a guardrail from misfiring, see [Guardrail fires on the wrong requests](/docs/protect/troubleshooting/guardrail-fires-on-the-wrong-requests).
+
+## See accuracy in Feedback
+
+Go to **Gateway** > **Guardrails** > **Feedback**. Everything submitted from request drawers rolls up there, in the **Feedback Summary by Check** card. It shows totals and accuracy per check, so you can see which checks your team is marking correct most often and which ones are drawing false positives or false negatives.
+
+## Dive deeper
+
+
+
+ Enable a check and set what it does when it fires
+
+
+ Diagnose and fix a check that's blocking or warning on requests it shouldn't
+
+
diff --git a/src/pages/docs/protect/guides/run-protect-from-the-sdk.mdx b/src/pages/docs/protect/guides/run-protect-from-the-sdk.mdx
new file mode 100644
index 00000000..41cfced0
--- /dev/null
+++ b/src/pages/docs/protect/guides/run-protect-from-the-sdk.mdx
@@ -0,0 +1,135 @@
+---
+title: "Run Protect from the SDK"
+description: "Screen a single input for toxicity, bias, prompt injection, or PII, straight from your own code."
+---
+
+This guide screens a single input by calling `protect()` directly from your own code: no dashboard involved. You can call it on incoming input before it reaches your model, or on the model's output before it reaches whoever receives it; `protect()` takes a string either way. The guide walks through initializing the client, building a rules list, running the check, and reading the result back, for both text and non-text input. For every parameter and return field, see the [Protect SDK reference](/docs/sdk/protect).
+
+## Initialize Protect
+
+Install the SDK first:
+
+```bash
+pip install ai-evaluation
+```
+
+```python
+from fi.evals import Protect
+
+protector = Protect()
+```
+
+Protect reads `FI_API_KEY` and `FI_SECRET_KEY` from your environment. Generate these on the [API Keys](/docs/admin-settings/api-keys) page if you don't have them yet.
+
+## Build the rules list
+
+`protect_rules` is a list of dicts, each naming one check to run with a `metric` key:
+
+```python
+rules = [
+ {"metric": "toxicity"},
+ {"metric": "bias_detection"},
+ {"metric": "prompt_injection"},
+ {"metric": "data_privacy_compliance"},
+]
+```
+
+For how these checks fit into the guardrail model, see [Understanding Protect](/docs/protect/concepts/understanding-protect).
+
+The SDK accepts these `metric` values: `toxicity`, `bias`, `bias_detection`, `sexist`, `prompt_injection`, `data_privacy_compliance`, and `pii`. A `metric` name outside this list raises an error, so confirm the name before you ship it.
+
+## Run the check
+
+Call `protector.protect()` with:
+
+- `inputs`: your input, as the first positional argument
+- `protect_rules`: the rules list
+- `action`: the message returned when a rule fails
+- `reason`: set `True` to include an explanation with the result
+- `timeout`: how long the check can run, in milliseconds
+
+```python
+text_to_check = "the text you want to screen"
+
+result = protector.protect(
+ text_to_check,
+ protect_rules=rules,
+ action="I'm sorry, I can't help you with that.",
+ reason=True,
+ timeout=25000,
+)
+```
+
+## Read the result
+
+`protect()` returns a dictionary shaped like this:
+
+```python
+{
+ "status": "failed",
+ "completed_rules": ["toxicity"],
+ "uncompleted_rules": ["bias_detection", "prompt_injection", "data_privacy_compliance"],
+ "failed_rule": ["toxicity"],
+ "messages": "I'm sorry, I can't help you with that.",
+ "reasons": ["Message contains content flagged as toxic."],
+ "time_taken": 0.42,
+}
+```
+
+- `status`: `"passed"` or `"failed"`
+- `messages`: on a failure, carries the `action` string instead of the original input
+- `completed_rules` / `uncompleted_rules`: which checks ran to completion, and which didn't; on a failure, the remaining checks can come back in `uncompleted_rules`
+- `failed_rule`: the check(s) that tripped a failure, as a list, for example `["toxicity"]`
+- `reasons`: holds an explanation for a failure when `reason=True` is set
+- `time_taken`: how long the check took, in seconds
+
+If a rule doesn't finish before `timeout` elapses, it lands in `uncompleted_rules` instead.
+
+Use `status` to decide what to send onward: on a failure, forward `messages` instead of the original input; on a pass, forward the input unchanged.
+
+```python
+if result["status"] == "failed":
+ response = result["messages"]
+else:
+ response = text_to_check
+```
+
+## Screen an image or audio input instead of text
+
+`inputs` isn't limited to text. Pass an image or audio path or URL in place of the text string, and call `protect()` the same way. The `rules` list carries over unchanged. For accepted URL and file formats, see the [Protect SDK reference](/docs/sdk/protect).
+
+```python
+result = protector.protect(
+ "/path/to/local/audio.wav",
+ protect_rules=rules,
+ action="Audio content cannot be processed",
+ reason=True,
+ timeout=25000,
+)
+```
+
+```python
+result = protector.protect(
+ "/path/to/local/image.png",
+ protect_rules=rules,
+ action="Image content cannot be processed",
+ reason=True,
+ timeout=25000,
+)
+```
+
+Text, image, and audio are the only accepted input types for this call. Image sets, PDFs, and knowledge bases are not accepted.
+
+## Dive deeper
+
+
+
+ Every protect() parameter and return field
+
+
+ What Protect checks and where it fits your pipeline
+
+
+ Apply checks as gateway guardrails on traffic configured in the dashboard
+
+
diff --git a/src/pages/docs/protect/guides/test-a-guardrail.mdx b/src/pages/docs/protect/guides/test-a-guardrail.mdx
new file mode 100644
index 00000000..c4bb7ec7
--- /dev/null
+++ b/src/pages/docs/protect/guides/test-a-guardrail.mdx
@@ -0,0 +1,55 @@
+---
+title: "Test a guardrail"
+description: "Run a prompt through your active guardrails from the Test tab and read the verdict."
+---
+
+The Test tab runs a single prompt through your active guardrail configuration and shows you the verdict immediately, so you can check how a guardrail behaves before any traffic from your app reaches it.
+
+
+You need a guardrail already turned on for a run to mean anything; with nothing configured, every prompt just passes through. If you haven't set one up yet, [turn on a guardrail](/docs/protect/guides/turn-on-a-guardrail) first.
+
+
+## Open the panel
+
+Go to **Gateway** > **Guardrails** > **Test**. That tab is the **Test Guardrails** panel.
+
+## Pick a prompt
+
+The panel gives you five example chips: **Safe prompt**, **PII test**, **Injection test**, **Secrets test**, and **Toxic content**. Click one to drop a ready-made prompt into the box, or skip the chips and type your own into "Enter a test prompt...".
+
+If you turned on a PII Detection guardrail using the PII example from Turn on a guardrail, click **PII test**. The chip's prompt carries the kind of data that guardrail is built to catch, so running it comes back blocked.
+
+## Name a model (optional)
+
+Below the prompt box, "Model (optional)" takes a model name, with placeholder text reading "e.g. gpt-4o-mini". It names the model the test request actually runs against.
+
+## Run the test
+
+Press **Run Test**. It stays disabled until there's a prompt in the box, whether you typed one or picked a chip, and while a test is in flight it reads "Running...".
+
+## Read the result
+
+When the test finishes, a result chip reports the outcome: "BLOCKED (446)" if a guardrail stopped the request, "WARNING (246)" if a guardrail flagged it without stopping it, or "OK" with the status code the call actually returned if it passed cleanly. See [Guardrail checks](/docs/protect/reference/guardrail-checks#response-statuses) for the status each action produces. For the PII test walkthrough above, expect "BLOCKED (446)".
+
+Below the chip, **Guardrail Headers** and **Response Body** show the guardrail headers and the response body the call returned. Open them when you're wiring your own client to react to blocked or warned responses and need the exact shape it will get, not just the summarized chip.
+
+## Check Recent Tests
+
+**Recent Tests** keeps your last 20 runs and clears when you reload the page.
+
+## If something doesn't look right
+
+**The test didn't run.** If the test call itself can't complete, an error alert replaces the result chip.
+
+**The verdict was wrong.** A chip that doesn't match what you expected, for example a prompt you thought would block coming back **OK**, points back to the guardrail's configuration rather than the panel. See [Guardrail fires on the wrong requests](/docs/protect/troubleshooting/guardrail-fires-on-the-wrong-requests) to fix it.
+
+## Dive deeper
+
+
+
+ Read a guardrail's trigger volume and latency, then correct a verdict it got wrong
+
+
+ Initialize the client, define checks, and run a check entirely from code
+
+
diff --git a/src/pages/docs/protect/guides/turn-on-a-guardrail.mdx b/src/pages/docs/protect/guides/turn-on-a-guardrail.mdx
new file mode 100644
index 00000000..3fef620e
--- /dev/null
+++ b/src/pages/docs/protect/guides/turn-on-a-guardrail.mdx
@@ -0,0 +1,57 @@
+---
+title: "Turn on a guardrail"
+description: "Enable a check in Protect's guardrail catalog and set how it behaves."
+---
+
+Every check in Protect's guardrail catalog starts out switched off. This guide flips one on end to end, using **PII Detection** configured for a support agent as the running example: enable it, configure it, and push it live. The same steps apply to any other check in the catalog.
+
+This guide assumes Agent Command Center, where Protect's guardrails live under Gateway, is already set up with at least one provider connected. If it isn't yet, start with the [Agent Command Center Quickstart](/docs/command-center/quickstart).
+
+## Find the check in Rules
+
+Go to **Gateway** > **Guardrails** > **Rules**. The Rules tab lists the full guardrail catalog as a list of check cards, split between **AI-Powered Checks** and **Rule-Based Checks**. **PII Detection** sits under **Rule-Based Checks**; find its card there. Each card carries a **Switch** and a pencil icon button.
+
+## Turn it on
+
+Click the **Switch** on the **PII Detection** card. This only stages the change: it's what raises the unsaved-changes banner covered in [Push the change to the gateway](#push-the-change-to-the-gateway) below, and nothing reaches the gateway until you save there. How the check behaves is set separately, from the pencil.
+
+## Set how it behaves
+
+Click the pencil icon button on the card to open the check's dialog, then set:
+
+1. **Enabled**: leave it on (it mirrors the card's **Switch**, so it already matches what you just set)
+2. **Action**: set to **Block**
+3. **Confidence Threshold**: leave at its default of 0.8
+
+For provider-backed checks, the dialog also has a Provider Settings section (see the note below). It always has Cancel and Save buttons. For what action and threshold actually do to a request, see [Understanding Protect](/docs/protect/concepts/understanding-protect).
+
+Click **Save** to close the dialog.
+
+
+If the check calls out to an external provider, **Provider Settings** is where that check's own provider connection lives (not the model provider connected during Agent Command Center setup). **PII Detection**'s dialog has no Provider Settings section, so skip it. (See [Guardrail checks](/docs/protect/reference/guardrail-checks) for which checks are provider-backed.) For a check that does use a provider, a credential you've already saved shows up masked; click into the field and it clears, ready for you to enter a new one.
+
+
+## Push the change to the gateway
+
+Nothing you set in the check's dialog reaches the gateway until you save here. Back on the Rules tab, an info banner now reads "You have unsaved changes. Save to push guardrail config to the gateway.", with **Reset** and **Save & Activate** buttons. This **Reset** discards the unsaved changes shown in the banner, not the check's saved configuration. It's a separate control from the **Reset** button that appears on a check's card once the check has been customized: that one clears the customization, switching the check back off and its action and threshold back to the defaults (**Block**, **0.8**). Click **Save & Activate**; while the save is in flight the button reads **Saving...**.
+
+## Confirm it in Overview
+
+Switch to the **Overview** tab. The row shows up as **pii-detector**, enabled, and its Stage column reads **Before LLM** (before any guardrail is enabled, this tab reads "No guardrails configured" instead).
+
+## Confirm the stage
+
+Click the row's **Edit** icon button to open **Edit Guardrail: pii-detector**. **Stage** already shows **Pre**, matching the **Before LLM** you just saw in Overview; this dialog is where you'd change it if you needed a different stage. The dialog also shows **Action** and **Threshold** (with the helper text "Numeric threshold for the guardrail (optional)"), both carried over from the pencil dialog: **Block** and **0.8**, plus **Cancel** and **Save** / **Saving...** buttons. Click **Save**. A toast confirms: `Guardrail "pii-detector" updated`.
+
+**PII Detection** is now enabled, blocking at a 0.8 confidence threshold, running at the **pre** stage.
+
+## Dive deeper
+
+
+
+ Send requests through the guardrail you just turned on and see how it responds
+
+
+ See what a guardrail has been catching since it went live
+
+
diff --git a/src/pages/docs/protect/index.mdx b/src/pages/docs/protect/index.mdx
index f6c46f6a..bbd94032 100644
--- a/src/pages/docs/protect/index.mdx
+++ b/src/pages/docs/protect/index.mdx
@@ -1,47 +1,44 @@
---
-title: "Future AGI Protect: Real-Time Safety and Policy Enforcement"
-description: "Future AGI Protect brings real-time safety and policy enforcement directly into your GenAI application flow to prevent harmful outputs."
+title: "Overview"
+description: "Where Protect's dashboard guardrails and SDK checks live"
---
-## About
-**Protect** is Future AGI's real-time guardrailing layer that screens every model input and output as it flows through your application. Unlike offline safety checks, Protect blocks or flags harmful content before it reaches end users, with no separate preprocessing pipeline needed.
+## What is Protect?
-It covers four safety dimensions:
+**Protect** is the safety layer over your AI traffic. It runs named guardrail checks, like PII detection or prompt injection, on each request and returns one of four actions: [block, warn, mask, or log](/docs/protect/concepts/understanding-protect).
-| Dimension | What it checks |
-|---|---|
-| **Content Moderation** | Toxicity, hate speech, threats, harassment, harmful language |
-| **Bias Detection** | Sexism, discrimination, harmful stereotypes |
-| **Security** | Prompt injection, adversarial manipulation, system prompt extraction |
-| **Data Privacy Compliance** | PII detection (names, emails, phone numbers, SSNs), GDPR/HIPAA violations |
+You turn guardrails on in the dashboard under Gateway → Guardrails, where they apply to traffic passing through a gateway. If your app isn't pointed at [Agent Command Center](/docs/command-center) yet, see the [quickstart](/docs/command-center/quickstart) for swapping in the base URL and API key. You can also call `protect()` directly from your code to run checks inline on text, image, and audio inputs. Use the dashboard if traffic goes through a gateway; call `protect()` directly if it doesn't.
-Built on Google's **Gemma 3n** foundation with specialized fine-tuned adapters, Protect operates natively across text, image, and audio modalities.
+## Start here
-
-
-## How Protect Connects to Other Features
-
-- **Agent Command Center**: Protect's safety dimensions can also be applied as guardrails in the Agent Command Center for all LLM traffic. [Learn more](/docs/command-center/features/guardrails)
-- **Evaluation**: The same safety checks (toxicity, bias, PII) are available as evaluation metrics for batch scoring across datasets. [Learn more](/docs/evaluation)
-- **Observability**: Protect results are logged as part of your traces, so you can see which requests were blocked and why. [Learn more](/docs/observe)
+
+
+ How guardrail checks, actions, and the two surfaces fit together
+
+
+ Enable a check on your gateway traffic from the dashboard
+
+
+ Call protect() directly from your code instead of the dashboard
+
+
+ Try a check against sample input before it goes live
+
+
+ Browse the first-party and external-provider checks available
+
+
+ Check trigger volume and correct a verdict from the request log
+
+
-## Getting Started
+## Related products
-
- Set up Protect and run your first safety check in minutes.
-
-
- Explore real-world use cases across content moderation, healthcare, security, and more.
+
+ Score the same kinds of risks across a dataset instead of live traffic
-
- Full SDK reference for the Protect module.
+
+ See what your AI app actually did, request by request
diff --git a/src/pages/docs/protect/reference/guardrail-checks.mdx b/src/pages/docs/protect/reference/guardrail-checks.mdx
new file mode 100644
index 00000000..be13aea8
--- /dev/null
+++ b/src/pages/docs/protect/reference/guardrail-checks.mdx
@@ -0,0 +1,140 @@
+---
+title: "Guardrail checks"
+description: "Every check name, configuration field, and response status Protect exposes, laid out as tables."
+---
+
+Checks run inside Protect's gateway pipeline and are configured from **Gateway > Guardrails**, either per-check or as pipeline-wide settings; see [Configuration fields](#configuration-fields) for where each field lives.
+
+## First-party checks
+
+These ten checks run inside Protect without an external provider. Three of them carry a different name on the **Gateway > Guardrails** settings tab.
+
+| Check | Also shown as |
+|---|---|
+| `pii-detector` | `pii-detection` |
+| `injection-detector` | `prompt-injection` |
+| `secrets-detector` | `secret-detection` |
+| `content-moderation` | Same |
+| `keyword-blocklist` | Same |
+| `topic-restriction` | Same |
+| `language-detection` | Same |
+| `system-prompt-protection` | Same |
+| `hallucination-detection` | Same |
+| `data-leakage-prevention` | Same |
+
+## Provider-backed checks
+
+Each of these 18 checks has a **Provider Settings** section in its Rules dialog (the per-check settings dialog under **Gateway > Guardrails**, covered in [Configuration fields](#configuration-fields)). Provider Settings is where that check's own provider configuration lives, not always a credential connection to an outside service.
+
+| Check |
+|---|
+| `futureagi-eval` |
+| `llama-guard` |
+| `azure-content-safety` |
+| `presidio-pii` |
+| `lakera-guard` |
+| `bedrock-guardrails` |
+| `hiddenlayer-guard` |
+| `aporia-guard` |
+| `pangea-guard` |
+| `dynamoai-guard` |
+| `enkrypt-guard` |
+| `ibm-ai-detector` |
+| `grayswan-guard` |
+| `lasso-guard` |
+| `crowdstrike-aidr` |
+| `zscaler-guard` |
+| `tool-permissions` |
+| `mcp-security` |
+
+## PII entities
+
+A `pii-detector` check (shown as `pii-detection` in the settings tab) can look for these 14 entity ids:
+
+| Entity id | Label |
+|---|---|
+| `SSN` | Social Security Number |
+| `CREDIT_CARD` | Credit Card Number |
+| `EMAIL` | Email Address |
+| `PHONE` | Phone Number |
+| `ADDRESS` | Physical Address |
+| `NAME` | Person Name |
+| `DOB` | Date of Birth |
+| `PASSPORT` | Passport Number |
+| `DRIVER_LICENSE` | Driver's License |
+| `IP_ADDRESS` | IP Address |
+| `BANK_ACCOUNT` | Bank Account Number |
+| `MEDICAL_RECORD` | Medical Record Number |
+| `AWS_KEY` | AWS Access Key |
+| `API_KEY` | API Key / Secret |
+
+## Topic categories
+
+A `topic-restriction` check groups its topics under 8 categories, most with their own subcategories; Custom ships with none.
+
+| Category id | Label | Subcategories |
+|---|---|---|
+| `violence` | Violence & Harm | weapons, self_harm, threats, graphic_violence |
+| `sexual` | Sexual Content | explicit, suggestive, minors |
+| `hate` | Hate Speech & Discrimination | racism, sexism, religious_hate, disability_hate |
+| `illegal` | Illegal Activities | drugs, fraud, hacking, terrorism |
+| `misinformation` | Misinformation | health_misinfo, political_misinfo, conspiracy |
+| `privacy` | Privacy Violations | doxxing, surveillance, stalking |
+| `profanity` | Profanity & Offensive Language | strong_profanity, slurs, insults |
+| `custom` | Custom Topics | none |
+
+## Configuration fields
+
+Most fields below apply to every check in both the [first-party](#first-party-checks) and [provider-backed](#provider-backed-checks) tables above. A check that carries provider fields additionally has a Provider Settings section holding them, which is why `keyword-blocklist` shows its Blocked Keywords there despite being first-party. Confidence Threshold appears on every check except `futureagi-eval` and `presidio-pii`.
+
+A check is configured from one of two dialogs: the per-check dialog covered under [Rules dialog](#rules-dialog), and the guardrail-level dialog covered under [Overview dialog](#overview-dialog). The two expose different fields: the Rules dialog offers Mask as an action option, plus a Confidence Threshold; the Overview dialog sets Stage.
+
+### Rules dialog
+
+| Field | Values | Default |
+|---|---|---|
+| Enabled | Toggle | On |
+| Action | Block, Warn, Mask, Log | Block |
+| Confidence Threshold | Slider from 0.0 to 1.0, marked at 0.0 / 0.5 / 1.0 | 0.8 |
+| Provider Settings (checks that have provider fields) | The check's provider configuration | |
+
+### Overview dialog
+
+| Field | Values | Default |
+|---|---|---|
+| Action | Block, Warn, Log | Block |
+| Stage | pre, post, both | pre |
+| Threshold | Numeric (optional) | |
+
+### Pipeline settings
+
+These apply to every check on the gateway rather than to an individual check.
+
+| Field | Values | Default | What it does |
+|---|---|---|---|
+| Mode | Parallel, Sequential | Parallel | Whether the gateway's checks run at the same time or one after another |
+| Fail Open | Toggle | On | What happens when a check doesn't return a verdict before Timeout runs out. On lets the request through unchecked; off applies the check's configured action instead |
+| Timeout | Milliseconds | 5000 ms | How long a check is given to return a verdict before Fail Open decides what happens next |
+
+## Response statuses
+
+A check that blocks or warns changes the gateway call's response status to one of these codes.
+
+| Status | Meaning |
+|---|---|
+| `403` | Blocked |
+| `446` | Blocked |
+| `246` | Warned |
+
+Both `403` and `446` indicate a blocked call. The condition that selects one over the other isn't documented here, so treat both as blocked when writing code that branches on status.
+
+## Keep exploring
+
+
+
+ Enable and configure a check on a gateway
+
+
+ Send a request through a check and see how it responds
+
+
diff --git a/src/pages/docs/protect/troubleshooting/guardrail-changes-not-taking-effect.mdx b/src/pages/docs/protect/troubleshooting/guardrail-changes-not-taking-effect.mdx
new file mode 100644
index 00000000..f14e388f
--- /dev/null
+++ b/src/pages/docs/protect/troubleshooting/guardrail-changes-not-taking-effect.mdx
@@ -0,0 +1,48 @@
+---
+title: "Guardrail changes not taking effect"
+description: "Diagnose a guardrail edit that isn't taking effect: an unsaved Rules tab banner, a failed enable toggle, or a Stage that doesn't cover where you're testing."
+---
+
+You changed a guardrail's action, threshold, or stage, but requests still come back exactly like they did before. Guardrails have two edit paths: the check cards on the **Rules** tab go through **Save & Activate**, while the **Overview** tab's Edit dialog saves on its own. Three causes account for almost every case, plus one run to confirm the fix landed.
+
+## Check for the unsaved-changes banner
+
+Go to **Gateway** > **Guardrails** > **Rules**. If an info banner sits above the check cards reading "You have unsaved changes. Save to push guardrail config to the gateway.", with **Reset** and **Save & Activate** buttons, your edit never went live.
+
+Click **Save & Activate**. If you've saved and behavior still hasn't changed, move on to the next check.
+
+## Check the guardrail's Stage
+
+If the banner is clear and the change definitely saved, go to **Gateway** > **Guardrails** > **Overview**, click the guardrail row's pencil icon button, and look at **Stage**. It's one of **pre**, **post**, or **both**.
+
+- Staged to **pre** but expecting it to catch something in the response? It never will: pre only looks at what goes in
+- Staged to **post** but expecting it to block the request itself? It never will either: post only looks at what comes back
+- Staged to **both**? It sees both sides, since **both** is the same check running once on the way in and once on the way out; see [Understanding Protect](/docs/protect/concepts/understanding-protect)
+
+Match **Stage** to the side of the exchange you actually need checked.
+
+## If the toggle failed
+
+If you flipped the **Enabled** switch on the guardrail's row in the **Overview** tab and saw the "Failed to toggle guardrail" error toast instead of a success toast reading "`` enabled", the enabled state itself never changed. Retry the toggle and don't move on until you see the success toast instead of the error toast.
+
+## Confirm with a Test tab run
+
+The fastest way to know whether any of these fixes worked, rather than waiting on real traffic, is one run in [**Gateway** > **Guardrails** > **Test**](/docs/protect/guides/test-a-guardrail). Send a prompt that should trigger the check and read the result chip in the Result card: it reads **BLOCKED (446)**, **WARNING (246)**, or **OK** with the status code, showing whether the guardrail fired on that prompt or let it through.
+
+## If the test still doesn't fire
+
+If the banner was clear, the toggle succeeded, the Stage already matched what you needed, and the Test tab run still shows the guardrail not firing, the cause isn't a leftover unsaved change or a stage mismatch. Contact support@futureagi.com with the guardrail's name, its Stage and Action, and the prompt you tested with.
+
+## Dive deeper
+
+
+
+ Enable a check and set its action, threshold, and stage
+
+
+ Fire a prompt at your guardrails and read the result chip
+
+
+ See how pre and post stages fit into request and response flow
+
+
diff --git a/src/pages/docs/protect/troubleshooting/guardrail-fires-on-the-wrong-requests.mdx b/src/pages/docs/protect/troubleshooting/guardrail-fires-on-the-wrong-requests.mdx
new file mode 100644
index 00000000..ead3f2f5
--- /dev/null
+++ b/src/pages/docs/protect/troubleshooting/guardrail-fires-on-the-wrong-requests.mdx
@@ -0,0 +1,63 @@
+---
+title: "Guardrail fires on the wrong requests"
+description: "Switch Action to Warn or Log while you tune, move Confidence Threshold, re-test the change, then record the verdict as feedback."
+---
+
+A guardrail check can go wrong in two directions: it blocks or warns on requests that were fine, or it lets through content it should have caught. Either way, the fix is the same set of levers, worked in order on the one check that's causing it.
+
+Go to **Gateway** > **Guardrails**. **Overview**, **Rules**, **Analytics**, **Feedback**, **Test**, and **Logs** are all tabs on that section; **Analytics** is where you diagnose which check is misfiring, and **Rules** is where you open that check's card to edit it.
+
+## Find the check that's firing
+
+Checks show up under **Rule** in the **Top Triggered Rules** table alongside **Triggers**, **Block**, and **Warn** counts. Open **Gateway** > **Guardrails** > **Analytics** to see it:
+
+- **Over-blocking:** a high Block count points to the check to look at first, not proof on its own that it's firing on requests that were fine
+- **Letting through:** if a specific kind of content, like PII or a prompt injection, is getting through, open the check meant to catch it
+
+## Switch Action to Warn or Log while you tune
+
+Switch **Action** from Block to Warn or Log before you touch the Confidence Threshold below, so a check you're still tuning doesn't stop real traffic. Go to **Gateway** > **Guardrails** > **Rules**, find the check's card, and click its pencil icon button to open the dialog (see [Turn on a guardrail](/docs/protect/guides/turn-on-a-guardrail) for the fuller walkthrough of setting these up). In the dialog's **Action** select:
+
+- **Block** stops the request, and counts toward that check's Block total in Top Triggered Rules
+- **Warn** lets the request through, and counts toward that check's Warn total in Top Triggered Rules
+- **Log** lets the request through too, and still counts toward Triggers, but the check's Block and Warn totals stay flat, for a quieter pass while you compare several changes
+
+**Mask** is also on the select, but it's a different use case outside this tuning flow. Set **Action** back to Block once you're satisfied with where the threshold lands.
+
+## Move its Confidence Threshold
+
+In the same dialog, find **Confidence Threshold**, a slider marked at 0.0, 0.5, and 1.0. Its untouched value is 0.8. Future AGI Eval and Presidio PII checks don't render this slider at all, so this lever isn't available for them. For what action and threshold actually do to a request, see [Understanding Protect](/docs/protect/concepts/understanding-protect).
+
+Raise the threshold and the check catches less, so move it up if the check is blocking or warning on requests that were fine. Lower it and the check catches more, so move it down if it's letting through content it should have caught. As a first move, try adjusting it by about 0.05 to 0.1, then re-test before adjusting further.
+
+Click **Save** in the dialog to stage the change, then back on the Rules tab, click **Save & Activate** on the "You have unsaved changes. Save to push guardrail config to the gateway." banner; the change doesn't reach the gateway until you do. If you re-test and nothing's changed, see [Guardrail changes not taking effect](/docs/protect/troubleshooting/guardrail-changes-not-taking-effect).
+
+## Re-test after each change
+
+After each change, switch to **Gateway** > **Guardrails** > **Test** (see [Test a guardrail](/docs/protect/guides/test-a-guardrail) for what it shows) and click the example chip that matches the case you're chasing, whether that's PII, an injection attempt, secrets, or toxic content, then run it. The result chip in the Result card reads BLOCKED (446), WARNING (246), or OK with the status code; that's the verdict to check against what you meant to happen. Re-running that chip after every threshold or Action change is how you confirm the change did what you meant, instead of stacking up several changes and losing track of which one mattered.
+
+## Record the verdict as feedback
+
+
+Feedback is a record, not a control. Submitting a False Positive or False Negative doesn't retune the check. Marking one here and skipping the Action and Confidence Threshold changes above leaves the check exactly as it was.
+
+
+From **Gateway** > **Guardrails** > **Logs**, open the request the check got wrong in its detail drawer. The drawer's **Guardrails** tab only appears when at least one check fired on that request; a request that nothing caught has no feedback controls at all, so it can't be marked False Negative from its own drawer. On the drawer's **Guardrails** tab, one set of feedback controls appears for each check that fired on that request; find the set for the check you're tuning and mark it **False Positive** if it fired on a request that was fine, or **False Negative** if it let through something it should have caught, then press **Submit Feedback**. That verdict rolls up into **Gateway** > **Guardrails** > **Feedback** alongside every other correction submitted for that check.
+
+## If it still fires wrong
+
+If the check still fires wrong after switching Action, moving the Confidence Threshold, and re-testing, the fix may be more than this check can offer on its own. Check whether a different check in the **Top Triggered Rules** table above is the actual source, or contact support@futureagi.com with the check's name, the Action and threshold you tried, and an example request it still gets wrong.
+
+## Dive deeper
+
+
+
+ Set a check's Action and Confidence Threshold for the first time
+
+
+ Run example or custom prompts through a check from the Test tab
+
+
+ Read Top Triggered Rules and the Feedback Summary by Check in full
+
+
diff --git a/src/pages/docs/protect/troubleshooting/protect-sdk-rejects-an-input.mdx b/src/pages/docs/protect/troubleshooting/protect-sdk-rejects-an-input.mdx
new file mode 100644
index 00000000..64d01d5b
--- /dev/null
+++ b/src/pages/docs/protect/troubleshooting/protect-sdk-rejects-an-input.mdx
@@ -0,0 +1,42 @@
+---
+title: "Protect SDK rejects an input"
+description: "Fix a protect() call that errors instead of returning a result"
+---
+
+Calling `protect()` fails instead of returning a dictionary with a status key, and nothing gets screened. Two causes produce this: an unsupported input type, or an unrecognized check name in `protect_rules`. If neither cause matches your situation, for example a missing API key when constructing `Protect()`, it's a different problem.
+
+## The input type isn't one Protect screens
+
+[Protect](/docs/protect/concepts/understanding-protect) screens text, image, and audio input, one at a time. Send it an image set, a PDF, or a knowledge base input, and the call fails before it screens anything.
+
+**Fix:** send one supported input per call.
+
+## The rule names a check protect() doesn't accept
+
+Each rule in `protect_rules` names a check with a `metric` key, and an unaccepted value fails the call before anything is screened. See [Run Protect from the SDK](/docs/protect/guides/run-protect-from-the-sdk) for the accepted `metric` values.
+
+**Fix:** match the `metric` value to one of the accepted names.
+
+```python
+# Fails before anything is screened
+rules = [{"metric": "jailbreak"}]
+
+# Fixed
+rules = [{"metric": "prompt_injection"}]
+```
+
+Confirm the exact spelling before you ship it, since a typo fails the same way as an unsupported name.
+
+## Dive deeper
+
+
+
+ Build the rules list and call protect() on text, image, and audio input
+
+
+ Every protect() parameter and return field
+
+
+ The separate set of checks configured as dashboard guardrails
+
+
diff --git a/src/pages/docs/prototype/concepts/understanding-prototype.mdx b/src/pages/docs/prototype/concepts/understanding-prototype.mdx
deleted file mode 100644
index 9e01aa1f..00000000
--- a/src/pages/docs/prototype/concepts/understanding-prototype.mdx
+++ /dev/null
@@ -1,51 +0,0 @@
----
-title: "Understanding Prototype: Pre-Production Testing in Future AGI"
-description: "Explains what Prototype is, the problem it solves, and how versions, traces, and evals work together before you ship to production."
----
-
-## About
-
-Prototype is a pre-production testing environment for LLM applications. It gives you a structured way to run multiple configurations of your application:different prompts, models, or parameters:and compare them on real outputs before deciding what goes to production.
-
-Without Prototype, the only way to know if a change made things better is to ship it and see. That means real users encounter regressions, hallucinations, or tone problems before you do. Prototype moves that discovery earlier: you run versions, score outputs automatically with evaluations, and compare everything in one dashboard before any version reaches production.
-
----
-
-## The core workflow
-
-1. **Register** your project with a version name and the evaluations you want to run.
-2. **Instrument** your application so every LLM call is automatically traced.
-3. **Run** your application:each generation is captured, tagged to its version, and scored.
-4. **Compare** versions in the Prototype dashboard by evaluation scores, cost, and latency.
-5. **Promote** the best-performing version to production.
-
-Every step is designed to be low-friction: instrumentation is automatic, scoring happens in the background, and the dashboard surfaces the comparison without manual analysis.
-
----
-
-## What gets measured
-
-Each version run is measured on three dimensions:
-
-| Dimension | What it captures |
-|---|---|
-| **Evaluation scores** | Quality metrics like context adherence, toxicity, hallucination detection, and tone:scored automatically on every generation. |
-| **Cost** | Token usage and estimated cost per generation for the model and configuration used. |
-| **Latency** | Response time per generation, so you can see the performance tradeoff of different models or prompts. |
-
-These three together give you a complete picture. A cheaper model may cost less but score worse on quality. A longer prompt may improve accuracy but add latency. Prototype shows all three at once.
-
----
-
-## Key concepts
-
-- **[Versions and Runs](/docs/prototype/concepts/versions-and-runs)**: What a version is and how runs get tagged and compared.
-- **[EvalTags and Mapping](/docs/prototype/features/evals)**: How evaluations attach to your runs and how span data maps to eval inputs.
-
----
-
-## Next steps
-
-- [Set Up Prototype](/docs/prototype/features/set-up-prototype): Register your project and instrument your app.
-- [Configure Evals for Prototype](/docs/prototype/features/evals): Define which evaluations run on your outputs.
-- [Choose Winner](/docs/prototype/features/choose-winner): Rank versions and promote the best to production.
diff --git a/src/pages/docs/prototype/concepts/versions-and-runs.mdx b/src/pages/docs/prototype/concepts/versions-and-runs.mdx
deleted file mode 100644
index 6ef69d8b..00000000
--- a/src/pages/docs/prototype/concepts/versions-and-runs.mdx
+++ /dev/null
@@ -1,56 +0,0 @@
----
-title: "Versions and Runs in Future AGI Prototype Testing"
-description: "What a version is in Prototype, how runs get tagged to a version, and how the dashboard uses versions to compare configurations."
----
-
-## About
-
-A version is a named configuration of your application: a specific prompt, model, or set of parameters. Every generation your instrumented application makes is tagged to the version it ran under, so the Prototype dashboard can group and compare them.
-
-Versions are how Prototype answers the question: "Is this new prompt actually better than the previous one?"
-
----
-
-## What a version is
-
-When you call `register()`, you pass a `project_version_name`. This name tags all traces produced by that registration to the same version. It can be anything meaningful: `gpt-4o-v1`, `shorter-system-prompt`, `with-few-shot-examples`.
-
-```python
-trace_provider = register(
- project_type=ProjectType.EXPERIMENT,
- project_name="my-chatbot",
- project_version_name="gpt-4o-concise-prompt",
-)
-```
-
-Every LLM call made after this registration is captured as a run under `gpt-4o-concise-prompt`.
-
----
-
-## What a run is
-
-A run is a single execution of your application under a version. Each run contains one or more spans: the LLM call, any retrieval steps, tool uses, or other instrumented operations. The spans carry the raw data:input messages, model response, token counts, cost, and latency.
-
-Runs are stored automatically. You do not need to manually log anything beyond registering and instrumenting your app.
-
----
-
-## Comparing versions
-
-To compare two configurations, register with different `project_version_name` values and run the same workload against each:
-
-| Version name | What changed |
-|---|---|
-| `baseline` | Original prompt, GPT-4o |
-| `shorter-prompt` | Condensed system message, GPT-4o |
-| `gpt-4o-mini` | Same prompt, cheaper model |
-
-The Prototype dashboard shows all versions for a project side by side, with evaluation scores, average cost, and average latency for each. The Choose Winner flow then lets you weight those metrics and rank the versions.
-
----
-
-## Next steps
-
-- [EvalTags and Mapping](/docs/prototype/features/evals): How evaluations score each run automatically.
-- [Set Up Prototype](/docs/prototype/features/set-up-prototype): Register your project and start capturing runs.
-- [Choose Winner](/docs/prototype/features/choose-winner): Rank versions by your chosen metrics and promote the best.
diff --git a/src/pages/docs/prototype/features/choose-winner.mdx b/src/pages/docs/prototype/features/choose-winner.mdx
deleted file mode 100644
index d90c6dae..00000000
--- a/src/pages/docs/prototype/features/choose-winner.mdx
+++ /dev/null
@@ -1,63 +0,0 @@
----
-title: "Choose Winner: Rank and Promote Best Prototype Version"
-description: "Rank prototype versions by evaluation scores, cost, and latency, then select and promote the best-performing version to production."
----
-
-## About
-
-When you have multiple versions of your application running in Prototype, you need a way to pick the best one. Choose Winner ranks all your versions based on the metrics that matter to you: evaluation scores, cost, and latency. You control how much each metric matters using sliders, and the platform calculates an overall score for each version. The highest-scoring version becomes the winner, and you can promote it to production directly from the dashboard, moving from prototype to production based on data instead of guesswork.
-
-{/* ARCADE EMBED START */}
-
-
-{/* ARCADE EMBED END */}
-
----
-
-## When to use
-
-- **Version comparison**: Compare multiple prompts, models, or parameter sets side by side on quality, cost, and latency before committing to one.
-- **Weighted ranking**: Prioritize what matters most for your use case (safety scores, response cost, or latency) and let the platform calculate the overall winner.
-- **Pre-production sign-off**: Make a documented, data-backed decision on which version to ship instead of relying on intuition.
-- **Seamless production promotion**: Promote the winning version directly from the dashboard with no code changes required.
-
----
-
-## How to
-
-
-
- Go to the [Prototype dashboard](https://app.futureagi.com/prototype) and open the project or experiment you want to compare.
- 
-
-
-
- Click the **Choose Winner** button to open the comparison and ranking flow.
- 
-
-
-
- Adjust the sliders for each metric (e.g. evaluation scores, cost, latency) to indicate how important they are on a scale from **0** (not important) to **10** (very important). Your choices determine how versions are ranked.
- 
-
-
-
- Based on the weights you set, all prototype versions are ranked. The version with the highest overall score is the winner. Select it to promote that configuration to production.
-
-
-
----
-
-## Next Steps
-
-
-
- Configure environment, register project, and instrument your app.
-
-
- Define which evals run on your prototype outputs.
-
-
- How Prototype fits in.
-
-
diff --git a/src/pages/docs/prototype/features/evals.mdx b/src/pages/docs/prototype/features/evals.mdx
deleted file mode 100644
index 28370397..00000000
--- a/src/pages/docs/prototype/features/evals.mdx
+++ /dev/null
@@ -1,193 +0,0 @@
----
-title: "Configure Evals for Prototype Testing in Future AGI"
-description: "Define which evaluations run on your prototype outputs using EvalTags, mapping, and optional custom evals in Future AGI Prototype."
----
-
-## About
-
-When running multiple versions of your application in Prototype, cost and latency alone don't tell you which version is better. Configuring evals adds quality scores to every run, so you can compare versions on what actually matters: does the output stay on topic, follow the right tone, avoid unsafe content, and answer accurately. Every generation is scored automatically, and the results appear in the Prototype dashboard alongside cost and latency so you can make a data-driven decision on which version to promote.
-
----
-
-## When to use
-
-- **Pre-production quality checks**: Score every run for hallucinations, tone, safety, or accuracy before promoting any version to production.
-- **Domain-specific criteria**: Use different evals depending on what matters for your use case.
-- **Reproducible scoring**: Same eval config across all versions so comparisons stay fair and consistent.
-- **Multi-version testing**: Run the same evals across all versions so rankings in the dashboard stay objective.
-
----
-
-## How to
-
-
-
- In your `register()` call, pass an `eval_tags` list (Python) or `evalTags` (TypeScript). Each tag specifies the eval name, span type and kind, mapping from your span attributes to the eval's required keys, optional custom display name, and the model to use.
-
-
-
- ```python Python
- eval_tags = [
- EvalTag(
- eval_name=EvalName.CONTEXT_ADHERENCE,
- type=EvalTagType.OBSERVATION_SPAN,
- value=EvalSpanKind.LLM,
- mapping={"context": "input.value", "output": "output.value"},
- custom_eval_name="context_check",
- model=ModelChoices.TURING_SMALL
- ),
- EvalTag(
- eval_name=EvalName.TOXICITY,
- type=EvalTagType.OBSERVATION_SPAN,
- value=EvalSpanKind.LLM,
- mapping={"input": "input.value"},
- custom_eval_name="toxicity_check",
- model=ModelChoices.TURING_SMALL
- )
- ]
- ```
-
- ```typescript JS/TS
- const evalTags = [
- new EvalTag({
- type: EvalTagType.OBSERVATION_SPAN,
- value: EvalSpanKind.LLM,
- eval_name: EvalName.CONTEXT_ADHERENCE,
- custom_eval_name: "context_check",
- mapping: { "context": "input.value", "output": "output.value" },
- model: ModelChoices.TURING_SMALL
- }),
- new EvalTag({
- type: EvalTagType.OBSERVATION_SPAN,
- value: EvalSpanKind.LLM,
- eval_name: EvalName.TOXICITY,
- custom_eval_name: "toxicity_check",
- mapping: { "input": "input.value" },
- model: ModelChoices.TURING_SMALL
- })
- ];
- ```
-
-
-
- | Field | Description |
- |-------|-------------|
- | `eval_name` | The evaluation to run. Must be a valid `EvalName` enum value. |
- | `type` | Where to apply the evaluation (e.g. `OBSERVATION_SPAN`). |
- | `value` | Kind of span to evaluate (e.g. `LLM`). |
- | `mapping` | Maps eval required keys to span attribute paths. [See below](#understanding-the-mapping-attribute). |
- | `custom_eval_name` | Display name for this eval in the dashboard. |
- | `model` | Model for Future AGI evals (e.g. `TURING_LARGE`, `TURING_SMALL`). |
-
-
-
- The `mapping` attribute connects eval requirements with your trace data. How it works:
-
- 1. **Each eval has required keys**: Different evals need different inputs (e.g. Context Adherence needs `context` and `output`).
- 2. **Spans have attributes**: Your spans (LLM, retriever, etc.) store data as key-value span attributes.
- 3. **Mapping connects them**: The mapping object specifies which span attribute to use for each required key.
-
- Example:
-
- ```python
- mapping={
- "context": "input.value",
- "output": "output.value"
- }
- ```
-
- - The eval's `context` key pulls from `input.value`: the raw input sent to the model.
- - The eval's `output` key pulls from `output.value`: the raw response from the model.
-
-
-
- `custom_eval_name` sets the display name shown in the Prototype dashboard for this eval. `eval_name` must always be a valid `EvalName` enum value: it selects which evaluation logic runs. Use `custom_eval_name` to give it a meaningful label for your project.
-
-
-
- ```python Python
- eval_tags = [
- EvalTag(
- eval_name=EvalName.CONTEXT_ADHERENCE,
- type=EvalTagType.OBSERVATION_SPAN,
- value=EvalSpanKind.LLM,
- mapping={"context": "input.value", "output": "output.value"},
- custom_eval_name="my_adherence_check",
- model=ModelChoices.TURING_SMALL
- ),
- ]
- ```
-
- ```typescript JS/TS
- const evalTags = [
- new EvalTag({
- type: EvalTagType.OBSERVATION_SPAN,
- value: EvalSpanKind.LLM,
- eval_name: EvalName.CONTEXT_ADHERENCE,
- custom_eval_name: "my_adherence_check",
- mapping: { "context": "input.value", "output": "output.value" },
- model: ModelChoices.TURING_SMALL
- })
- ];
- ```
-
-
-
-
-
-
-For the full list of built-in evals and their required mapping keys, see [Built-in evals](/docs/evaluation/builtin).
-
-
----
-
-## Span Attribute Paths Reference
-
-For OpenAI (and most LLM instrumentors), the standard span attribute paths are:
-
-| Data | Span attribute path |
-|---|---|
-| System message content | `gen_ai.input.messages.0.message.content` |
-| User message content | `gen_ai.input.messages.1.message.content` |
-| Model response | `gen_ai.output.messages.0.message.content` |
-
-The index (`.0.`, `.1.`) corresponds to the position of the message in the messages array passed to the model.
-
----
-
-## Required Keys by Eval
-
-Each eval has its own required mapping keys. Common ones:
-
-| Eval | Required keys |
-|---|---|
-| Context Adherence | `context`, `output` |
-| Toxicity | `input` |
-| Completeness | `input`, `output` |
-| Detect Hallucination | `input`, `output` |
-| Prompt Injection | `input` |
-| Tone | `input` |
-
-For the full list, see [Built-in evals](/docs/evaluation/builtin).
-
----
-
-## Next Steps
-
-
-
- Configure environment, register project, and instrument your app.
-
-
- Rank versions and promote the best to production.
-
-
- How Prototype fits in.
-
-
- Running evals and built-in eval details.
-
-
- Full list of 70+ built-in evals with mapping keys and output types.
-
-
diff --git a/src/pages/docs/prototype/features/set-up-prototype.mdx b/src/pages/docs/prototype/features/set-up-prototype.mdx
deleted file mode 100644
index ccb7b8d7..00000000
--- a/src/pages/docs/prototype/features/set-up-prototype.mdx
+++ /dev/null
@@ -1,192 +0,0 @@
----
-title: "Set Up Prototype: Configure Your Future AGI Prototype App"
-description: "Configure your environment, register your prototype project, and instrument your app so traces and evals appear in the Prototype dashboard."
----
-
-## About
-
-Prototype lets you run multiple versions of your AI application side by side — different prompts, models, or parameters — and compare them on real outputs before deciding what goes to production. Setting up Prototype is how you bring your application into that environment.
-
-You register your project with a version name, instrument your application so its LLM calls are automatically traced, and optionally attach evaluations so each run is scored. From that point, every generation your app makes is captured in the Prototype dashboard under the version it belongs to, ready to compare against other versions by quality, cost, and latency.
-
----
-
-## When to use
-
-- **First-time prototype**: Get your project and version registered and start sending traces so you can compare different prompts or models.
-- **Comparing versions**: Use `project_version_name` (or equivalent) so each run is tagged and comparable in the dashboard.
-- **Eval-ready setup**: Register with optional `eval_tags` so prototype outputs are scored (e.g. tone, safety) without changing code later.
-- **Framework integration**: Use Auto Instrumentor for OpenAI (or manual tracing) so existing LLM calls are automatically traced.
-
----
-
-## How to
-
-
-
- Install the core instrumentation package and the framework instrumentor for your LLM provider.
-
-
-
- ```bash Python
- pip install fi-instrumentation-otel traceAI-openai
- ```
-
- ```bash JS/TS
- npm install @traceai/fi-core @traceai/openai
- ```
-
-
-
-
-
- Set environment variables so your app can talk to Future AGI. Get your API keys [here](https://app.futureagi.com/dashboard/keys).
-
-
-
- ```python Python
- import os
- os.environ["FI_API_KEY"] = "YOUR_API_KEY"
- os.environ["FI_SECRET_KEY"] = "YOUR_SECRET_KEY"
- ```
-
- ```typescript JS/TS
- process.env.FI_API_KEY = "YOUR_API_KEY";
- process.env.FI_SECRET_KEY = "YOUR_SECRET_KEY";
- ```
-
-
-
-
-
- Call `register()` with your project name, version name (for comparing runs), and optional eval tags. Use `ProjectType.EXPERIMENT` for prototyping.
-
-
-
- ```python Python
- from fi_instrumentation import register, Transport
- from fi_instrumentation.fi_types import ProjectType, EvalName, EvalTag, EvalTagType, EvalSpanKind, ModelChoices
-
- trace_provider = register(
- project_type=ProjectType.EXPERIMENT,
- project_name="FUTURE_AGI",
- project_version_name="openai-exp",
- transport=Transport.HTTP,
- eval_tags=[
- EvalTag(
- eval_name=EvalName.TONE,
- value=EvalSpanKind.LLM,
- type=EvalTagType.OBSERVATION_SPAN,
- model=ModelChoices.TURING_LARGE,
- mapping={"input": "llm.input_messages"},
- custom_eval_name="",
- ),
- ],
- )
- ```
-
- ```typescript JS/TS
- import { register, Transport, ProjectType, EvalName, EvalTag, EvalTagType, EvalSpanKind, ModelChoices } from "@traceai/fi-core";
-
- const evalTag = await EvalTag.create({
- type: EvalTagType.OBSERVATION_SPAN,
- value: EvalSpanKind.LLM,
- eval_name: EvalName.CHUNK_ATTRIBUTION,
- custom_eval_name: "Chunk_Attribution",
- mapping: { "context": "raw.input", "output": "raw.output" },
- model: ModelChoices.TURING_SMALL
- });
-
- const tracerProvider = register({
- projectName: "FUTURE_AGI",
- projectType: ProjectType.EXPERIMENT,
- transport: Transport.HTTP,
- projectVersionName: "openai-exp",
- evalTags: [evalTag]
- });
- ```
-
-
-
- | Property (Python) | Property (TypeScript) | Description |
- |------------------------|----------------------|-------------|
- | `project_type` | `projectType` | Use `ProjectType.EXPERIMENT` for Prototype. |
- | `project_name` | `projectName` | Your project name. |
- | `project_version_name` | `projectVersionName` | (optional) Version id for this prototype so you can compare runs. |
- | `eval_tags` | `evalTags` | (optional) Evals to run on prototype outputs. [Learn more](/docs/prototype/features/evals) |
- | `transport` | `transport` | (optional) `GRPC` or `HTTP`. Defaults to `HTTP`. |
-
-
- Python uses **snake_case**; TypeScript uses **camelCase** for these properties.
-
-
-
-
- Use one of:
-
- - **Auto Instrumentor**: Recommended; use Future AGI's instrumentor for your framework (e.g. OpenAI).
- - **Manual tracing**: OpenTelemetry for custom setups.
-
- **Example: OpenAI (Auto Instrumentor):** Instrument your client after registering. Traces will appear in the [Prototype dashboard](https://app.futureagi.com/dashboard/projects/experiment).
-
-
-
- ```python Python
- from traceai_openai import OpenAIInstrumentor
- import openai
-
- OpenAIInstrumentor().instrument(tracer_provider=trace_provider)
-
- client = openai.OpenAI()
- completion = client.chat.completions.create(
- model="gpt-4o",
- messages=[{"role": "user", "content": "Write a one-sentence bedtime story about a unicorn."}]
- )
- print(completion.choices[0].message.content)
- ```
-
- ```typescript JS/TS
- import { OpenAIInstrumentation } from "@traceai/openai";
- import { registerInstrumentations } from "@opentelemetry/instrumentation";
- import { OpenAI } from "openai";
-
- registerInstrumentations({
- instrumentations: [new OpenAIInstrumentation({})],
- tracerProvider: tracerProvider
- });
-
- const client = new OpenAI({ apiKey: process.env.OPENAI_API_KEY });
- const completion = await client.chat.completions.create({
- model: "gpt-4o",
- messages: [{ role: "user", content: "Write a one-sentence bedtime story about a unicorn." }]
- });
- console.log(completion.choices[0].message.content);
- ```
-
-
-
- For more frameworks and options, see the Auto Instrumentation docs.
-
-
-
- After setting up your prototype, you can:
- - **Configure evals**: Define which evaluations run on your prototype outputs (EvalTag, mapping, model). [Configure evals for prototype](/docs/prototype/features/evals)
- - **Compare and choose winner**: Rank versions by evals, cost, and latency, then promote the best. [Choose winner](/docs/prototype/features/choose-winner)
-
-
-
----
-
-## Next Steps
-
-
-
- Define EvalTags and mapping for your runs.
-
-
- Rank versions and select the best to promote.
-
-
- How Prototype fits in.
-
-
diff --git a/src/pages/docs/prototype/index.mdx b/src/pages/docs/prototype/index.mdx
deleted file mode 100644
index 32af4b1c..00000000
--- a/src/pages/docs/prototype/index.mdx
+++ /dev/null
@@ -1,36 +0,0 @@
----
-title: "Future AGI Prototype: Test and Compare LLM Configurations"
-description: "Test and compare LLM configurations, prompts, and parameters in Future AGI Prototype before deploying changes to production."
----
-
-## About
-
-Prototype is Future AGI's pre-production testing environment for AI applications. When you change a prompt, switch models, or adjust how your AI behaves, you need a way to verify the change actually improves things before it reaches real users. Without a structured testing step, teams either ship blind or run informal tests that don't reflect real usage, and find out something is wrong only after it has caused problems.
-
-Prototype solves this by letting you run multiple versions of your application side by side against real inputs. Each version is traced and scored automatically using evaluations you define: output quality, tone, safety, factual accuracy, or any custom criteria. Once you have results, the Prototype dashboard shows all versions compared by eval scores, cost, and latency. You use the Choose Winner flow to set how much each metric matters, let the platform rank the versions, and promote the best one to production.
-
-
-
----
-
-## How Prototype Connects to Other Features
-
-- **Evaluation**: Prototype uses the same eval templates as the rest of the platform. Scores from 70+ built-in metrics are calculated automatically per version. [Learn more](/docs/evaluation)
-- **Observability**: Every prototype run is traced. After promoting a winner, traces continue in Observe so you monitor production performance. [Learn more](/docs/observe)
-- **Optimization**: Use prototype results to identify which prompt to optimize further. [Learn more](/docs/optimization)
-
----
-
-## Getting Started
-
-
-
- Configure your environment, register your project, and instrument your app so traces appear in the dashboard.
-
-
- Define which evaluations run on your prototype outputs using EvalTags, mapping, and model selection.
-
-
- Rank prototype versions by eval scores, cost, and latency, then promote the best to production.
-
-
diff --git a/src/pages/docs/quickstart/annotations.mdx b/src/pages/docs/quickstart/annotations.mdx
index 883da8cf..2d951157 100644
--- a/src/pages/docs/quickstart/annotations.mdx
+++ b/src/pages/docs/quickstart/annotations.mdx
@@ -82,10 +82,10 @@ In this walkthrough you will create an annotation label, set up a queue, add tra
## Next Steps
-
+
Explore all five label types and their configuration options.
-
+
Configure assignment strategies, multi-annotator requirements, and review workflows.
diff --git a/src/pages/docs/quickstart/generate-synthetic-data.mdx b/src/pages/docs/quickstart/generate-synthetic-data.mdx
index bddee517..ef0b8bd9 100644
--- a/src/pages/docs/quickstart/generate-synthetic-data.mdx
+++ b/src/pages/docs/quickstart/generate-synthetic-data.mdx
@@ -62,5 +62,5 @@ description: "Generate synthetic datasets with Future AGI. Define schemas, colum
- [Run evaluations on your dataset](/docs/evaluation) to test AI outputs against the generated data
- [Use Knowledge Base](/docs/knowledge-base) to ground synthetic data generation with your own documents
-- [Run prompts on your dataset](/docs/dataset/features/run-prompt) to add model-generated columns
-- [Set up experiments](/docs/dataset/features/experiments) to compare different prompts or models against your dataset
+- [Run prompts on your dataset](/docs/dataset/guides/run-a-prompt-on-every-row) to add model-generated columns
+- [Set up experiments](/docs/dataset/guides/run-an-experiment) to compare different prompts or models against your dataset
diff --git a/src/pages/docs/quickstart/prompts.mdx b/src/pages/docs/quickstart/prompts.mdx
index b41b6e83..7a0eb8b8 100644
--- a/src/pages/docs/quickstart/prompts.mdx
+++ b/src/pages/docs/quickstart/prompts.mdx
@@ -135,6 +135,6 @@ description: "Create and manage AI prompts in Future AGI's Prompt Workbench. Des
## Next Steps
-- [Use prompts via SDK](/docs/prompt/features/sdk) to serve and manage prompts programmatically in your application
+- [Use prompts via SDK](/docs/prompt/reference/sdk-api) to serve and manage prompts programmatically in your application
- [Optimize your prompts](/docs/optimization) to automatically improve prompt performance using evaluation-driven feedback
- [Run evaluations](/docs/evaluation) to measure how well your prompts perform across different inputs
diff --git a/src/pages/docs/quickstart/running-evals-in-simulation.mdx b/src/pages/docs/quickstart/running-evals-in-simulation.mdx
index 72eb3e81..8c542ffa 100644
--- a/src/pages/docs/quickstart/running-evals-in-simulation.mdx
+++ b/src/pages/docs/quickstart/running-evals-in-simulation.mdx
@@ -9,7 +9,7 @@ description: "Run evaluations in Future AGI simulations to test AI agents agains
---
-**Prerequisites:** Before starting, make sure you have set up your [Agent Definition](/docs/simulation/concepts/agent-definition), [Scenarios](/docs/simulation/concepts/scenarios), and [Personas](/docs/simulation/concepts/personas).
+**Prerequisites:** Before starting, make sure you have set up your [Agent Definition](/docs/simulation/concepts/agent-definitions), [Scenarios](/docs/simulation/concepts/scenarios), and [Personas](/docs/simulation/concepts/personas).
@@ -105,5 +105,5 @@ If the built-in evals don't cover your use case, you can create your own.
## Next Steps
- [Browse all built-in evals](/docs/evaluation/builtin) to find metrics that fit your use case
-- [Set up agent definitions](/docs/simulation/concepts/agent-definition) if you haven't already
+- [Set up agent definitions](/docs/simulation/concepts/agent-definitions) if you haven't already
- [Learn about simulation concepts](/docs/simulation) for a deeper understanding of how scenarios and personas work
diff --git a/src/pages/docs/sdk/annotation-queues/index.mdx b/src/pages/docs/sdk/annotation-queues/index.mdx
index f2e07bde..9d785f69 100644
--- a/src/pages/docs/sdk/annotation-queues/index.mdx
+++ b/src/pages/docs/sdk/annotation-queues/index.mdx
@@ -3,7 +3,7 @@ title: "AnnotationQueue class"
description: "Reference for the AnnotationQueue class in the Future AGI Python SDK: create, fetch, and populate annotation queues programmatically."
---
-For step-by-step examples, see the [Annotation Queue Using SDK](/docs/annotations/sdk/annotation-queue-using-sdk) guide.
+For step-by-step examples, see the [Annotation Queue Using SDK](/docs/annotations/reference/sdk-api) guide.
The `AnnotationQueue` class is the SDK client for managing annotation queues, items, scores, and analytics. Annotation queues let you organize traces, sessions, datasets, and simulation outputs for structured human review. You can define custom labels, set how many annotations are needed per item, and add guidelines to keep feedback consistent.
diff --git a/src/pages/docs/sdk/tracing/annotating-using-api.mdx b/src/pages/docs/sdk/tracing/annotating-using-api.mdx
index 631da906..c8d0a3ff 100644
--- a/src/pages/docs/sdk/tracing/annotating-using-api.mdx
+++ b/src/pages/docs/sdk/tracing/annotating-using-api.mdx
@@ -26,7 +26,7 @@ Traces show what happened but not whether the result was correct, helpful, or sa
- Annotation labels must be created before using the API. See the [Labels guide](/docs/annotations/features/labels) for how to create and configure labels (text, numeric, categorical, star, thumbs up/down).
+ Annotation labels must be created before using the API. See the [Labels guide](/docs/annotations/reference/label-types-and-values) for how to create and configure labels (text, numeric, categorical, star, thumbs up/down).
diff --git a/src/pages/docs/simulation/concepts/agent-definition.mdx b/src/pages/docs/simulation/concepts/agent-definition.mdx
deleted file mode 100644
index 0d40352b..00000000
--- a/src/pages/docs/simulation/concepts/agent-definition.mdx
+++ /dev/null
@@ -1,146 +0,0 @@
----
-title: "Agent Definition: AI Behavior Config in Simulations"
-description: "An agent definition configures how your AI agent behaves during voice or chat conversations in Future AGI simulation tests."
----
-
-## About
-
-An **agent definition** is the configuration record for a single AI agent in Simulate. It describes which agent is being tested and how the platform connects to it.
-
-Each agent definition includes:
-- A **name** and **type** (voice or chat)
-- **Connection details**: provider (e.g. Vapi, Retell), assistant ID, and API key
-- For voice agents: contact number, inbound/outbound setting, and language(s)
-- For custom/WebSocket agents: websocket URL and headers
-- Optional **knowledge base** and **observability provider**
-
-Agent definitions support **versioning**. Each version stores a snapshot of the configuration so you can run simulations against a specific version, compare versions, or roll back. Scenarios and run tests reference an agent definition (and a chosen version) to execute tests against that agent.
-
-## Creating Agent Definition
-
-
-
- Open **Simulate** from the sidebar and click **Agent Definition**.
-
-
- Fill in the required basic information.
-
- 
-
- | Field | Description |
- |-------|-------------|
- | Agent Type | Choose **Voice** or **Chat**. Voice agents are used for phone or voice-channel simulations; chat agents for chat-based ones. |
- | Agent Name | A unique, descriptive name for your agent. This name appears when you select an agent in scenarios and run tests, and is used for the observability project name if you enable observability. |
- | Language | The primary language (or multiple languages) the agent will use. Select one or more from the supported list (e.g. English, Spanish, French). This drives language-specific behavior in simulations. |
-
- If you already have an assistant configured in a provider (e.g. Vapi or Retell), you can use **Sync from provider** (see the next steps) to pull the assistant’s name and prompt into the form after entering the provider, API key, and assistant ID.
-
-
- Configure how the platform connects to your agent. This section is required for outbound agents and for syncing or running tests.
-
- 
-
- | Field | Description |
- |-------|-------------|
- | Voice/Chat Provider | The provider that hosts your agent (e.g. **Vapi**, **Retell**, **Eleven Labs**, or **Others** for custom/WebSocket). See [supported providers](/future-agi/docs/integrations/overview#voice) for setup. |
- | Assistant ID | The assistant or agent ID from your provider’s dashboard. **Required when connection type is Outbound.** |
- | API Key | Your provider API key for authentication. **Required when connection type is Outbound.** |
- | Observability Provider | Enable observability to track calls and performance. When enabled, a project is created in [Observe](https://app.futureagi.com/dashboard/observe) under your agent’s name. |
-
- For **Outbound** agents, both **Assistant ID** and **API Key** must be set; otherwise saving will fail with a validation error.
-
-
- If your agent is already set up in **Vapi** or **Retell**, you can pull the assistant’s name and system prompt into the form. Enter the **Voice/Chat Provider**, **Assistant ID**, and **API Key**, then use the sync action. The platform retrieves the assistant name and prompt and fills the corresponding fields. If the API key or assistant ID is wrong, you will see an error.
-
-
-
- Describe what the agent does and optionally attach a knowledge base.
-
- 
-
- - **Description / model:** Add a description of the agent’s purpose and, if needed, set the **model** and **model details** (e.g. system prompt, personality). This is snapshotted when you create a version. If you used “Sync from provider,” the prompt may already be filled.
- - **Knowledge base (optional):** A knowledge base is the **source of truth** your agent is expected to know (FAQs, SOPs, product docs, compliance policies). Attaching one lets evals check whether the agent’s responses match your real content, catching wrong answers or off-policy responses.
-
-
- Learn more in the [Knowledge base overview](/docs/knowledge-base).
-
-
-
- Configure contact and call direction (for voice agents).
-
- - **Contact number:** The phone number the agent will use.
- - **Country code:** Select the country code for the contact number.
- - **Connection type:**
- - **Inbound (ON):** The agent receives incoming calls from customers.
- - **Outbound (OFF):** The agent places calls to customers. For Outbound, **Assistant ID** and **API Key** (in Agent configuration) must be set.
-
-
- When saving, provide a **commit message** to track changes. The system creates a new version with a snapshot of the current configuration.
-
-
- Turn this on to track your agent’s performance. After you enable it and run a test, a project is created in your agent’s name in the [Observe](https://app.futureagi.com/dashboard/observe) section.
-
-
-
-## Agent Detail View
-
-After creating an agent, open it from the list to access the detail screen. Here you can edit the configuration, manage versions, and view results.
-
-
-
-- **Agent select dropdown**: Switch between agents without leaving the page.
-- **Version management (left)**: All versions for this agent, newest first. Click a version to load it.
-- **Create new version**: Opens a drawer to create a new version from the current config.
-
-
-
- View and edit the agent’s definition. Shows the same fields used during creation.
- 
-
- - **Basic information**: Agent name, type, and language(s).
- - **Provider and connection**: Voice/Chat provider, Assistant ID, API key, observability provider.
- - **Behavior**: Description, model, model details, and optional knowledge base.
- - **Contact (voice agents)**: Contact number, country code, and connection type.
-
- Saving creates a **new version** with a snapshot of the updated config. Previous versions remain in the version list. You can also **delete** the agent from this tab.
-
-
- Each version is a saved snapshot of your agent’s configuration. A version has a version number, status (**Draft**, **Active**, **Archived**, **Deprecated**), and a commit message. Only one version can be **Active** at a time; run tests use the active version by default.
-
-
-
- Click **Create new version**. Enter a **Commit message**, update fields if needed, then click **Save**.
- 
- Use clear commit messages (e.g. "Updated system prompt for support flow") so version history stays useful.
-
-
- Click a version in the list on the left. The main area loads that version’s config. Saving from here creates a new version.
- 
- Switching only changes what you are viewing; it does not delete other versions.
-
-
- Use **Activate** on a version to make it the default for run tests. The previously active version remains in the list.
-
-
- Use **Restore** to revert the agent definition to an older snapshot. You can then save as a new version.
-
-
- Use **Delete** to soft-delete a version. You cannot delete the only active version; activate another version first.
-
-
-
-
- After running simulations, you can view performance analytics and call logs for each agent version. See [View Results](/docs/simulation/features/view-results) for details.
-
-
-
-## Next Steps
-
-
-
- Create scenarios (graph, script, or dataset-backed) for the user journey or test cases.
-
-
- Tie your agent to scenarios, attach evals, and run the simulation.
-
-
diff --git a/src/pages/docs/simulation/concepts/agent-definitions.mdx b/src/pages/docs/simulation/concepts/agent-definitions.mdx
new file mode 100644
index 00000000..768cb3e8
--- /dev/null
+++ b/src/pages/docs/simulation/concepts/agent-definitions.mdx
@@ -0,0 +1,70 @@
+---
+title: "Agent definitions & versions"
+description: "The voice or chat agent you simulate against, and how versions keep runs comparable"
+---
+
+
+## What an agent definition holds
+
+An **agent definition** is the record of the agent you simulate against in a [simulation](/docs/simulation). It carries:
+
+- a name, and a type of voice or chat
+- the connection details Simulation needs to reach the agent
+- an optional [knowledge base](/docs/knowledge-base) the agent draws on
+
+Every edit you make can be frozen as a numbered **version**, and a run points at a definition and at one of its versions, so it names exactly which configuration it exercised.
+
+
+
+
+
+*The agent definition is your side of the conversation: what Simulation reaches, and how*
+
+## Voice and chat agents
+
+The type decides how Simulation reaches your agent, so a definition is wired one of two ways. [Connect your agent](/docs/simulation/guides/connect-your-agent) walks through both.
+
+### Voice
+
+A voice agent is reached over the phone, and a number is all Simulation needs. You give the definition one, the Future AGI caller dials it to hold the conversation, and the agent on the other end can run on any provider you like.
+
+Connecting **Vapi** or **Retell** as the provider is optional, and goes further. Those two are integrated natively, so if your agent runs on one of them the definition can also carry:
+
+- an assistant ID and an API key for that provider
+- the assistant's name and system prompt, pulled straight from the provider so the definition matches what runs in production
+- a concurrency limit that caps how many calls run at once
+
+### Chat
+
+A chat agent is answered by your own code. Simulation hands your service each turn the [persona](/docs/simulation/concepts/personas) says, through the [SDK](/docs/simulation/reference/sdk-api), and your agent replies until the conversation ends. No phone number is involved, so a chat agent needs no provider connection.
+
+## What a version captures
+
+A definition holds one live configuration, the one you edit. A **version** freezes it: creating a version snapshots that configuration under a number, and you write a commit message describing what changed, the same way you would for code. The newest version becomes the **active** one, and activating a version archives every other version of that definition, so exactly one is active at a time.
+
+A run executes against the snapshot a version holds rather than whatever the definition looks like today. If you don't pick a version when you set up a run, it uses the definition's latest version, so the configuration a run exercises is always one you can name afterwards. Older versions stay runnable, which is how you re-run a configuration you have since edited past.
+
+
+Editing a definition doesn't create a version. It changes the live configuration and leaves existing versions untouched, so create one whenever you want the configuration you just ran preserved. [Connect your agent](/docs/simulation/guides/connect-your-agent) covers this alongside the setup.
+
+
+## Versions keep runs comparable
+
+Every run records the version it ran against, and a version's results are the [evaluation](/docs/evaluation) scores from the conversations that ran against it. So you can change the agent, create a version, and see whether the scores moved against a frozen baseline instead of a shifting one. If a new version regresses, the one before it is still there to run.
+
+## Keep exploring
+
+
+
+ The situations you run the agent through
+
+
+ The customer your agent talks to in a run
+
+
+ How a version's scores and pass rates are produced
+
+
+ Wire up a voice or chat agent step by step
+
+
diff --git a/src/pages/docs/simulation/concepts/global-nodes.mdx b/src/pages/docs/simulation/concepts/global-nodes.mdx
deleted file mode 100644
index 44ca8bfc..00000000
--- a/src/pages/docs/simulation/concepts/global-nodes.mdx
+++ /dev/null
@@ -1,90 +0,0 @@
----
-title: "Global Nodes: Mid-Flow Conversation Steps in Simulations"
-description: "A global node is a conversation step the agent can enter at any point. Use it for off-topic questions, interrupts, and 'talk to a human' requests."
----
-
-## About
-
-A **global node** is a special kind of conversation node. The agent can jump into it at **any point** in the call or chat, not just when the flow reaches it.
-
-Most nodes only run when an edge points to them. A global node works differently. When the user says something that matches what the node is for, the agent drops into the global node, no matter where in the flow they were. The node's **prompt** is what tells the agent both what to do and when to enter it.
-
-Think of it like a shortcut. The other nodes are stops on a track. A global node is a button the user can press at any stop.
-
-## When to use a global node
-
-Use a global node when something can come up at any moment in the conversation, and you do not want to draw an arrow from every node to handle it. Common examples:
-
-- **Off-topic questions.** The user asks something that is not part of the main flow, and the agent needs to answer it the same way every time.
-- **Interrupts.** The user suddenly asks about pricing, refunds, or hours while the agent is doing something else.
-- **Talk to a human.** The user asks for an agent or a manager. One global node handles this no matter where they are.
-- **Small repeated tasks.** Asking for a callback number, confirming consent, or reading a disclosure that can happen at any stage.
-
-## Global node vs regular conversation node
-
-| Property | Regular conversation node | Global node |
-| -------- | ------------------------- | ----------- |
-| How it is reached | By following an edge from another node. | The agent jumps in when the user's message matches what the prompt describes. |
-| Must be reachable from start? | Yes. | No. |
-| Needs an incoming edge? | Yes (unless it is the start node). | No. |
-| Needs an outgoing edge? | Yes. It must route to another node or to an `endCall` / `transferCall` tool. | No. |
-| Good for | A specific step in the flow. | Things that can happen at any point (off-topic, interrupt, transfer). |
-
-Only conversation nodes can be global. Tool nodes like `endCall` and `transferCall` cannot.
-
-## Enabling Global on a node
-
-You turn a node into a global node from the workflow editor. A conversation node has two fields in the side panel: **Prompt** and the **Enable Global Node** toggle.
-
-
-
- In the sidebar, open **Simulate** and click **Scenarios**. Open your scenario and click **Edit graph** to open the editor.
-
- 
-
-
- Click the conversation node you want to make global. A side panel opens with the node's settings. If you do not have a suitable node yet, drag a new **Conversation** node from the palette first.
-
- 
-
-
- In the **Prompt** field, describe both the situation that should bring the agent here and what the agent should do. The first sentence is effectively the "when", and the rest is the "what".
-
- Example: *"The user has asked to speak to a human. Confirm that they want to speak to a human, and ask what they would like to talk to the human about."*
-
- Keep the opening of the prompt specific so the node only fires when it should. Vague openings like "the user is confused" fire too often and get in the way.
-
-
- Turn on **Enable Global Node** ("Make this node available from any point in the conversation"). Once it is on, the node shows a **Global** chip in the graph, so you can tell it apart from normal nodes.
-
- 
-
-
- If the node should end in a tool (for example, transferring the call or ending it), drag an edge from the global node to an `endCall` or `transferCall` tool node. This step is optional. With no outgoing edge, control returns to the main flow after the global node runs.
-
-
- Click **Save flow** in the Flow Builder sidebar. The global node is now live. It will fire whenever the user's message matches what its prompt describes.
-
- 
-
-
-
-## How global nodes behave in a simulation
-
-- On every turn, the simulator looks at what the user just said and checks it against each global node's prompt. If one matches, the agent jumps into that global node, no matter which node the flow was on.
-- Global nodes do not need to be connected to the rest of the graph. The normal rules ("every node must be reachable", "every node needs an incoming edge") do not apply to them.
-- You can still connect a global node with edges if you want. For example, you can point it at an `endCall` or `transferCall` tool, so the call ends or transfers after the global node runs.
-- Do not make most of your nodes global. If everything is global, the flow has no structure and the agent can jump around at random. Use global nodes for exceptions, not the main path.
-
-Only conversation nodes can be global. Tool nodes always follow the edges you draw.
-
-## Next Steps
-
-
-
- See how scenarios, graphs, and nodes fit together in Simulate.
-
-
- Run a scenario that uses global nodes against your agent.
-
-
diff --git a/src/pages/docs/simulation/concepts/optimization.mdx b/src/pages/docs/simulation/concepts/optimization.mdx
new file mode 100644
index 00000000..2235c702
--- /dev/null
+++ b/src/pages/docs/simulation/concepts/optimization.mdx
@@ -0,0 +1,56 @@
+---
+title: "Optimization"
+description: "Let an algorithm search for a better prompt when hand-fixing isn't enough"
+---
+
+## What optimization is
+
+**Optimization** rewrites your agent's prompt automatically, scored by the same [evals](/docs/evaluation) your simulation runs. Rather than hand-editing the prompt and rerunning, you let an algorithm propose many candidate prompts, score each one, and hand you the best. The prompt is what a run changes, few-shot examples included; your [agent definition](/docs/simulation/concepts/agent-definitions) and your evals stay as they are.
+
+Reach for it when hand-fixing has stalled. [Fix My Agent](/docs/simulation/guides/fix-my-agent) is the lighter first move: it reads a finished run and hands you a prioritised list of issues to fix yourself. An optimization run goes further and does the rewriting for you.
+
+## How an optimization run works
+
+An optimization run starts from a [simulation run](/docs/simulation/concepts/runs-and-results) you have already completed. It samples the conversations recorded in that run, and every candidate prompt is scored against that same sample, so the comparison holds still while the search moves.
+
+conversations + eval scores"] -->|"frozen sample"| SCORE
+ subgraph SEARCH["The search loop"]
+ direction LR
+ ALG["Algorithm"] -->|"proposes"| CAND["Candidate prompt"]
+ CAND --> SCORE["Trial score"]
+ SCORE -->|"guides the next round"| ALG
+ end
+ SEARCH --> BEST["Best prompt"]
+`} />
+
+Say your refund agent keeps failing a resolution eval. You point an optimization run at the simulation run where it failed, the algorithm generates candidate prompts, each is scored on that same eval against those conversations, and the best-performing prompt surfaces for you to review and apply. Because the score is your own eval, the winner is the prompt that best satisfies the bar you set.
+
+## The algorithms
+
+You pick the search strategy. They differ in how hard they search and in what they change, and searching harder costs more model calls. Start with Random Search for a baseline, then match the pick to what's wrong:
+
+- **Random Search** tries simple variations, the cheapest way to see how much room a prompt has
+- **Bayesian** keeps your wording and searches over which few-shot examples and settings work best, so reach for it when the prompt reads fine but the examples feel arbitrary
+- **ProTeGi** critiques each failure and applies a targeted fix, keeping several candidate revisions in play at once, for a prompt that is mostly right
+- **Meta-Prompt** analyses failures and rewrites the whole prompt through deeper reasoning, for a prompt that needs rethinking rather than patching
+- **PromptWizard** mutates the prompt across different thinking styles, then critiques and refines the top performers
+- **GEPA** runs an evolutionary search across generations of candidates, the widest search of the six
+
+The [Optimization](/docs/optimization) product docs cover each algorithm in depth.
+
+## Keep exploring
+
+
+
+ Start an optimization run on a finished simulation
+
+
+ Read the trials and apply the winning prompt
+
+
+ Get a prioritised list of fixes to apply by hand
+
+
diff --git a/src/pages/docs/simulation/concepts/personas.mdx b/src/pages/docs/simulation/concepts/personas.mdx
index 0a0346fa..410e04fe 100644
--- a/src/pages/docs/simulation/concepts/personas.mdx
+++ b/src/pages/docs/simulation/concepts/personas.mdx
@@ -1,120 +1,57 @@
---
-title: "Personas: Simulated Customer Profiles in Future AGI"
-description: "Personas in Future AGI represent the customers or users your AI agent interacts with in simulation. Define them to create realistic test conversations."
+title: "Personas"
+description: "Give your test customer a personality, a voice, and a way of talking"
---
-## About
-
-A **persona** defines who the simulated customer is during a test — their demographics, personality, and communication style. The simulator uses these traits to play the "customer" side of the conversation, making interactions feel realistic rather than scripted. You can use one of **18 pre-built personas** or create your own. Personas are typed as **voice** or **chat**, each with settings specific to that channel.
-
-
-
-## Voice vs Chat
-
-When creating a custom persona, you select whether it is for **voice** or **chat** simulations.
-- **Voice**: Used in phone or voice-channel tests. You can set speech-related options: **accent**, **conversation speed**, **background noise**, and **interrupt sensitivity** (how easily the persona can be interrupted or can interrupt). Behavioural and basic-information settings apply to both.
-- **Chat**: Used in text-based tests. You can set text-related options: **tone**, **verbosity**, **punctuation style**, **emoji usage**, **slang**, **typos frequency**, and **regional mix**. Basic information and behavioural settings apply to both.
-
-The same persona is not shared across voice and chat; create separate personas if you need both for the same “type” of customer.
-
-## Creating a custom persona
-
-From the Personas page, click **Create your own persona**. Choose **Voice** or **Chat** depending on whether the persona will be used in voice or chat simulations; the form then shows the relevant settings for that type. Follow the steps in the tab that matches your choice.
-
-
-
-
- Use this flow when creating a persona for **voice** (phone or voice-channel) simulations.
-
-
-
- Enter the core details the simulator uses to identify this persona. Same fields as chat; all optional except name and description if required by the UI.
-
- 
-
- | Property | Description |
- | -------- | ----------- |
- | Persona name | Name you give this persona (e.g. "Price-sensitive caller"). |
- | Description | Short description, e.g. "An angry customer who is not happy with the service." |
- | Gender (optional) | One or more: Male, Female. |
- | Age (optional) | One or more ranges: 18–25, 25–32, 32–40, 40–50, 50–60, 60+. |
- | Location (optional) | One or more: United States, Canada, United Kingdom, Australia, India. |
-
-
- Set **personality traits** and **communication style**. For voice, also set **accent** (e.g. American, Australian, Indian, Neutral). This controls how the persona responds and speaks during the call.
-
- 
-
-
- Control how the voice conversation runs: **conversation speed** (e.g. very slow, slow, moderate, fast, very fast), how the simulator responds, and **background noise** (on/off) for realism. You can also set **interrupt sensitivity** and **finished-speaking sensitivity** so the persona behaves naturally with turn-taking and interruptions.
-
- 
-
-
- Add any extra attributes not covered by the predefined fields (e.g. "insurance_type", "objection_pattern") so scenarios can reference them.
-
- 
- 
-
-
- Add free-form instructions (e.g. "Always ask for a supervisor after the first objection.").
-
- 
-
-
- Click **Add** (or **Save**). The persona appears in your list and can be used in voice scenarios and run tests.
-
-
-
-
- Use this flow when creating a persona for **chat** simulations.
-
-
-
- Enter the core details the simulator uses to identify this persona. Same fields as voice; all optional except name and description if required by the UI.
-
- 
-
- | Property | Description |
- | -------- | ----------- |
- | Persona name | Name you give this persona (e.g. "Price-sensitive buyer"). |
- | Description | Short description, e.g. "An angry customer who is not happy with the service." |
- | Gender (optional) | One or more: Male, Female. |
- | Age (optional) | One or more ranges: 18–25, 25–32, 32–40, 40–50, 50–60, 60+. |
- | Location (optional) | One or more: United States, Canada, United Kingdom, Australia, India. |
-
-
- Set **personality traits** and **communication style**. These define how the persona types and responds in chat.
-
- 
-
-
- Control how the chat persona writes: **tone** (formal, casual, neutral), **verbosity** (brief, balanced, detailed), **punctuation style** (clean, minimal, expressive, erratic), **emoji usage** (never, light, regular, heavy), **slang usage**, **typos frequency**, and **regional mix**. These make the text feel realistic for the persona.
-
- 
-
-
- Add any extra attributes not covered by the predefined fields (e.g. "insurance_type", "objection_pattern") so scenarios can reference them.
-
- 
- 
-
-
- Add free-form instructions (e.g. "Always ask for a supervisor after the first objection.").
-
- 
-
-
- Click **Add** (or **Save**). The persona appears in your list and can be used in chat scenarios and run tests.
-
-
-
-
-
-## Next Steps
+
+## What a persona is
+
+Two agents are in play during a run, and they sit on opposite sides of the conversation. Your [agent definition](/docs/simulation/concepts/agent-definitions) is the one under test. A **persona** is the customer facing it: their demographics, personality, and communication style. The simulator plays the persona so a run feels like a real interaction instead of a fixed script. "The Frustrated Subscriber", for example, is a voice persona who speaks fast, interrupts often, and pushes back on every answer.
+
+Personas are the pool your [scenarios](/docs/simulation/concepts/scenarios) draw from. You pick the personas a scenario should cover when you create it, and each of the scenario's rows, the individual test cases it holds, carries those traits into the run. **A persona reaches a run through its scenario, so there is no persona to pick at launch.**
+
+
+
+
+
+*A list of personas feeds the scenarios your agent is put through*
+
+## What you can customize
+
+Every persona is typed **voice** or **chat**, fixed when you create it, and the lists you pick from are filtered to match the simulation you're building, so a voice run only ever offers voice personas. From there a persona is customizable across a wide surface, from who the customer is to exactly how they sound. Set as much or as little as you need, and every field falls back to a sensible default:
+
+- **Basic info**: name, description, and demographics, meaning gender, age range, location, and profession
+- **Behaviour**: personality traits and communication style
+- **Voice traits**, on a voice persona: accent, the languages spoken and whether the persona is multilingual, conversation speed, background sound, and turn-taking, split into interrupt sensitivity and finished-speaking sensitivity
+- **Chat traits**, on a chat persona: tone, verbosity, punctuation style, emoji usage, slang, typo frequency, and regional mix
+- **Custom properties**: attributes you name yourself, like `objection_pattern` or `insurance_type`, which travel with the persona into the [scenario](/docs/simulation/concepts/scenarios) rows generated from it
+- **Instructions**: free-form guidance the simulator always follows, like "ask for a supervisor after the first objection"
+
+The [Built-in personas](/docs/simulation/reference/built-in-personas) reference lists all 18 alongside every field and the values it accepts.
+
+## Built-in and custom personas
+
+Future AGI ships 18 built-in personas you can use as-is, from "The Confused First-Time User" to "The No-Nonsense Executive" to "The Enterprise IT Admin". When none of them fit, you [create a custom persona](/docs/simulation/guides/create-personas) in your workspace, shape it across the layers above, and reuse it across every run.
+
+## The simulator agent
+
+The **simulator agent** is the actor on the customer side: the model that generates the customer's turns, the voice it speaks in, and the pacing it keeps. Future AGI runs it, and it isn't yours to configure.
+
+The persona is how you shape it. Everything you'd otherwise want to tune about the actor, how fast it talks, how readily it interrupts, how it comes across, you set on the persona instead, and the simulator agent plays what it finds there. That is what personas are for.
+
+## Keep exploring
-
- Tie your agent and scenario to a run test, attach evals, and run the simulation.
+
+ All 18, and every field they can set
+
+
+ Build and reuse a custom persona
+
+
+ What comes back once the persona has played its part
+
+
+ Turn a real call into a persona-driven test
diff --git a/src/pages/docs/simulation/concepts/replay.mdx b/src/pages/docs/simulation/concepts/replay.mdx
new file mode 100644
index 00000000..50bd50d4
--- /dev/null
+++ b/src/pages/docs/simulation/concepts/replay.mdx
@@ -0,0 +1,61 @@
+---
+title: "Replay"
+description: "Reproduce a real production bug as a test, instead of guessing at a synthetic one"
+---
+
+## What replay is
+
+**Replay** takes a real production conversation from [Observe](/docs/observe) and rebuilds it as two things you can rerun: a [scenario](/docs/simulation/concepts/scenarios) made from the transcript, and an [agent definition](/docs/simulation/concepts/agent-definitions) carrying the configuration that call ran on. Say a caller got quoted the wrong refund window last week: you pick that exact conversation, rebuild it, and rerun it to confirm your fix holds. **Instead of guessing at a synthetic case, you reproduce the real one.**
+
+
+Replay reads what Observe already recorded, so your app has to be sending traces before any of this is available. Replaying a whole conversation additionally needs those traces to carry a session ID, and voice replay needs [voice observability](/docs/observe/concepts/voice-observability) on the original call.
+
+
+## Session replay and trace replay
+
+You choose how much of the production data becomes one conversation.
+
+### Session
+
+A whole session, every trace under one `session_id` in order, replays as a single multi-turn conversation. Reach for it when you want to rerun full production conversations end to end.
+
+### Trace
+
+Each selected trace replays as its own one-turn conversation, an input and an output. Reach for it when you want to replay individual calls or single-turn interactions.
+
+## Chat and voice
+
+Replay works in both channels, and voice carries more of the original setup across.
+
+### Chat replay
+
+Rebuilds the conversation from the production transcripts and runs it against your agent.
+
+### Voice replay
+
+Goes further: it pulls the original voice setup (system prompt, assistant settings, and provider config) from the production call, so the replayed call runs on the same configuration as the original.
+
+Vapi is the one config extraction is built around. Retell and Bland.ai calls replay too, though what you get back to compare afterwards is the transcript rather than the full call. If your calls run on any other stack, voice replay can't reconstruct the original configuration, so replay the conversation as chat instead.
+
+## What a replay produces
+
+The recreated agent definition reproduces the agent as it behaved in production, which makes it **your baseline, not your fix**. You edit it, or point the run at a newer version, and the difference between the two runs is what your change did.
+
+
+
+Once the run finishes, you compare the replayed conversation with the original side by side, transcripts, metrics, and for voice the audio, so you can see exactly what your change moved.
+
+## A replayed failure becomes a regression test
+
+A production failure that only happened once can slip away. A replayed scenario is an ordinary [scenario](/docs/simulation/concepts/scenarios), so keep it alongside the others you run on every change and the exact conversation that broke becomes something every future version has to pass.
+
+## Keep exploring
+
+
+
+ Turn a production session into a chat simulation, step by step
+
+
+ Rerun a production call on its original voice configuration
+
+
diff --git a/src/pages/docs/simulation/concepts/runs-and-results.mdx b/src/pages/docs/simulation/concepts/runs-and-results.mdx
new file mode 100644
index 00000000..e9905b8e
--- /dev/null
+++ b/src/pages/docs/simulation/concepts/runs-and-results.mdx
@@ -0,0 +1,60 @@
+---
+title: "Runs & results"
+description: "How a simulation executes, and what each run leaves behind to read and compare"
+---
+
+## What a run is
+
+A **run test** bundles everything one simulation needs: an [agent version](/docs/simulation/concepts/agent-definitions), the [scenarios](/docs/simulation/concepts/scenarios) to play, and the [evals](/docs/evaluation) that score the result. The [personas](/docs/simulation/concepts/personas) come along inside the scenarios, attached when each one was built, so you don't pick them again here. Run one version of your support agent against a refund scenario carrying a frustrated persona, scored by a resolution eval, and you have one run test.
+
+You create and start one from the dashboard: [Run a voice simulation](/docs/simulation/guides/run-voice-simulation) and [Run a chat simulation](/docs/simulation/guides/run-chat-simulation) walk through it end to end.
+
+## From run to calls
+
+Starting a run test creates an **execution**, one attempt at playing the whole bundle, and the execution fans out into calls: one call per scenario, or one per row of the scenario's table. Each call runs its conversation end to end.
+
+one agent version, the scenarios, the evals"] --> EX(["Execution"])
+ EX --> C1["Call refund, frustrated caller"]
+ EX --> C2["Call refund, polite regular"]
+ EX --> C3["Call booking change"]
+ C1 --> REC["Each call leaves a transcript, metrics, and eval results"]
+ C2 --> REC
+ C3 --> REC
+`} />
+
+An execution reports where it is. It moves through **pending**, **running**, and **evaluating**, then finishes as **completed**, or as **failed** or **cancelled** when it stops early; **cancelling** is the brief state while a stop you requested takes effect.
+## The results each call carries
+
+Every call is the record of one conversation. It holds:
+
+- The **transcript**, turn by turn, with the speaker role on each turn
+- The **recording**, for voice calls
+- **Conversation metrics**, like latency, talk ratio, and cost
+- **Eval results**, a score per metric for that call, and **tool-call results** when the run test was created with tool evaluation switched on
+
+The exact metric fields and speaker roles live in the [Call metrics](/docs/simulation/reference/call-metrics) reference.
+
+## Reruns and snapshots
+
+From a run's results view, covered in [Explore results](/docs/simulation/guides/explore-results), you can rerun a whole call or only its evals, for example after changing an eval config. A rerun doesn't overwrite the earlier result: the previous run is snapshotted, so you can compare a call before and after a change instead of losing the baseline.
+
+## Runs are comparable
+
+Because a run pins an agent version and replays the same scenarios, two runs differ only by what you changed. That makes their scores directly comparable, so you can trend quality across versions and catch a regression before it ships.
+
+## Keep exploring
+
+
+
+ Read a run: calls, transcripts, and analytics
+
+
+ Every metric a call carries, and what each speaker role means
+
+
+ Turn failing runs into an improved agent
+
+
diff --git a/src/pages/docs/simulation/concepts/scenarios.mdx b/src/pages/docs/simulation/concepts/scenarios.mdx
index 89391ec9..d2b0c436 100644
--- a/src/pages/docs/simulation/concepts/scenarios.mdx
+++ b/src/pages/docs/simulation/concepts/scenarios.mdx
@@ -1,213 +1,66 @@
---
-title: "Scenarios: Test Cases and Conversation Flows in Simulations"
-description: "Scenarios defines the test cases, customer profiles, and conversation flows that your AI agent will encounter during simulations."
+title: "Scenarios"
+description: "The test case a simulation runs: a situation, a flow, and the rows that fill it"
---
-
-
-## About
-
-A **scenario** is a structured test case that simulates real-world interactions your agent will face in Simulate. Each scenario includes **personas** (who the customer is), **situations** (context and circumstances), and **outcomes** (expected results and success criteria). You can create scenarios manually or use automatic generation. Run tests use scenarios to drive voice or chat simulations against your agent so you can measure performance and improve over time.
-
-## Creating a scenario
-
-Navigate to **Simulate** → **Scenarios** → **Add scenario**, then choose how you want to create the scenario. The platform supports four types; pick the one that fits your use case.
-
-
-
- In the sidebar, open **Simulate** and click **Scenarios**. You’ll see the list of existing scenarios. Click **Add scenario** (or equivalent) to create a new one.
- 
-
-
- Select one of the four scenario types. Each type has a different way of defining test cases and conversation flows.
- 
-
-
-
-**Choose your scenario type:**
-
-
-
- **What it is:** Build or auto-generate conversation flows with a visual graph. Best for comprehensive test suites with multiple paths and branches.
-
- **Option A: Auto-generate**
-
-
- Enable **Auto Generate Graph**, then select **Agent definition**, set **Number of rows**, and provide a **Scenario description**.
-
-
- Click **Generate**. The system creates conversation paths, personas, situations, and outcomes automatically.
- 
-
-
-
- **Option B: Manual graph**
-
-
- Add nodes from the palette: **Conversation** (purple, start/continue conversations), **End call** (red, end or branch), **Transfer call** (orange, transfer or merge paths).
-
-
- Connect nodes with edges, then click each node to configure prompts, messages, and conditions.
-
-
- Save the graph. It will be used when you run tests with this scenario.
- 
-
-
-
-
- **What it is:** Use a table (CSV, Excel, or synthetic data) to define many test cases. Good when you have or want structured customer profiles and variables.
-
-
-
- Select **Dataset** as the scenario type.
- 
-
-
- **Upload** a file, **use a sample dataset**, or **Generate synthetic data** (number of records, demographics, insurance types, objection patterns, etc.).
-
-
- Map columns to scenario variables if needed. Save to create the scenario.
-
-
- Learn how to create [synthetic datasets](/docs/dataset/concept/synthetic-data).
-
-
- **What it is:** Paste or upload a call script (customer and agent lines). The system builds a graph and generates personas, situations, and outcomes from the script. Ideal for compliance or specific dialogue tests.
-
-
-
- Select **Upload Script** (or Script) as the scenario type. Choose **Agent definition**, set **Number of rows**, and add a **Scenario description**.
-
-
- Paste or upload your **Script content** (e.g. TXT, DOCX, PDF). Use lines like Customer: ... and Agent: ...; you can add [EXPECTED: ...] for outcomes.
- 
-
-
- Save; the system parses the script into nodes and generates scenario rows.
-
-
-
-
- **What it is:** Define a Standard Operating Procedure (steps and expectations). The system turns the SOP into a graph and scenario rows. Good for consistency, compliance, and training.
-
-
-
- Select **Call / Chat SOP** as the scenario type. Choose **Agent definition**, **Number of rows**, and **Scenario description**.
-
-
- Enter **SOP content** as numbered steps (e.g. Greeting and verification → Incident details → Assessment and next steps).
- 
-
-
- Save; the system builds the graph and scenarios from the SOP.
-
-
-
-
-
-## After creating: Scenario detail view
-
-When you open a scenario from the list, you see the **scenario detail** screen. Here you can view the graph (if any), the scenario table, and the simulator prompt; edit the graph or prompt; and add or remove rows. Use this view to refine test cases before running tests.
-
-
-
-**Layout:**
-
-- **Scenario list / breadcrumb**: Navigate back to the scenario list or switch scenarios.
-- **Graph (if applicable)**: Visual workflow; use **Edit graph** to change nodes and connections.
-- **Scenario table**: Rows = test cases; columns = variables (persona, situation, outcome, etc.). Use **Add rows** or delete selected rows to change the set of test cases.
-- **Simulator prompt**: The prompt used to drive the simulator; use **Edit prompt** to change it and reference row columns with {`{{column_name}}`}.
-
-
-
- **What you see:** The detail view shows the **graph** (for workflow/script/SOP scenarios), the **scenario table** (all rows and columns), and the **simulator prompt** used when running tests. The graph summarizes the conversation flow; the table holds the concrete test cases (personas, situations, outcomes); the prompt tells the simulator how to behave and can pull values from the table via {`{{column_name}}`}.
-
- **What you can do:** Use this as the home view to understand the scenario, then switch to **Edit graph**, **Edit prompt**, or **Scenario table** tabs to make changes.
-
-
- **What it is:** The workflow editor for scenarios that have a graph. You can add, delete, and edit nodes and change connections between them.
- 
-
- **Process:**
-
-
-
- On the scenario detail page, click **Edit graph**. The interactive workflow editor opens with the current graph.
-
-
- Add nodes from the palette (Conversation, End call, Transfer call), delete nodes you don’t need, and drag edges to connect or reconnect nodes. Click a node to edit its configuration (prompts, messages, conditions).
-
-
- Save your changes. The updated graph is used when you run tests with this scenario.
-
-
-
-
-
- **What it is:** The **simulator prompt** controls how the simulator agent behaves during the test. You can reference scenario table columns so each row gets personalized behavior (e.g. {`{{customer_name}}`}, {`{{objection_type}}`}).
-
- **What you see:** In the prompt editor, variables that match a column in the scenario table are highlighted (e.g. green = column exists; red = column missing and should be added or generated). Ensure every variable you use in the prompt exists as a column in the scenario table.
-
- 
-
-
- Click **Edit** on the prompt section.
-
-
- Change the prompt text; use {`{{column_name}}`} to insert values from the scenario table.
-
-
- Fix any red (missing) variables by adding the corresponding column to the scenario table or adjusting the variable name.
-
-
- Save your changes.
- 
-
-
-
-
- **What it is:** The scenario table lists all test cases (rows). Each row is one run in a test; columns are variables (persona attributes, situation, outcome, etc.) that you can use in the simulator prompt. You can add rows or delete selected rows.
-
- **Add rows: process:**
-
-
-
- Click **Add rows** on the scenario detail page.
- 
-
-
- - **From existing dataset or experiment**: Pick a dataset, map its columns to the scenario columns, and add the rows.
- 
- - **Generate using AI**: Enter a prompt; the system generates new rows based on it.
- 
- - **Add empty rows**: Add blank rows and fill them in manually.
- 
-
-
- Complete the flow (mapping, prompt, or count) and confirm. New rows appear in the scenario table.
-
-
-
- **Delete rows:** Select rows using the checkboxes, then use the delete action. Selected rows are removed from the scenario.
- 
-
-
-
-## Next Steps
-
-
-
- Define personas or simulator agents that play the "customer" side in the scenario.
+
+## What a scenario is
+
+A **scenario** is one test case for your [agent definition](/docs/simulation/concepts/agent-definitions): the situation a conversation starts from and the flow it should follow. A refund request, a booking change, a billing dispute, each is a scenario the simulator can run.
+
+
+
+
+
+A run picks one scenario from your library and plays it out against your agent. The rest of this page is what a single scenario holds.
+
+## The flow
+
+The **flow** is the path the conversation is meant to take, drawn as a graph. Each step is a **node**, and the edges between them are the routes the conversation can follow:
+
+| Node | What it does |
+|---|---|
+| **Conversation** | A step where the agent and customer talk, the ordinary building block of a flow |
+| **End call** / **End chat** | Terminates the conversation on that branch |
+| **Transfer call** / **Transfer chat** | Hands off, typically to a human, and can merge paths |
+
+You never draw the flow by hand unless you want to. It comes out of whichever source you built the scenario from, a workflow graph, a dataset, a script, or an SOP, and [Create scenarios](/docs/simulation/guides/create-scenarios) covers all four.
+
+### Global nodes
+
+A **global node** is a Conversation node the agent can reach at any point, not only when the flow arrives at it. It fits anything that can interrupt at any moment: an off-topic question, a sudden pricing query, a "talk to a human" request. One global node covers that case from anywhere in the flow, so you don't draw an edge to it from every step. Only Conversation nodes can be global.
+
+## One scenario, many conversations
+
+A scenario is not a single fixed script. It carries a table of rows, and each row plays out as its own conversation through the same flow. That is how one refund request scenario becomes a hundred concrete tests instead of one.
+
+A row pairs a [persona](/docs/simulation/concepts/personas) with the details of that particular case:
+
+| Persona | Amount | Objection |
+|---|---|---|
+| Frustrated caller | $240 | Wants an exception to the 30-day window |
+| Polite regular | $35 | Confused about the refund timeline |
+
+The personas come from the ones you attach when you build the scenario, which is why the persona never has to be chosen again at run time: it is already in the row.
+
+## The simulator prompt
+
+Every scenario carries a **simulator prompt**: the instructions the simulator follows to play the customer. It draws on the current row, so the same prompt produces a different customer for every row while the situation the scenario describes stays fixed.
+
+## Scenarios are reusable
+
+A run pins the scenarios it used, so the set is recorded with the run rather than re-read later. Point the same set at a new agent version and the comparison is honest: the test didn't move, the agent did. Kept together, they become a regression suite, and any conversation that used to pass and now fails stands out.
+
+## Keep exploring
+
+
+
+ Build one from a workflow graph, a dataset, a script, or an SOP
+
+
+ The customer that plays out the scenario
-
- Tie your agent and scenario to a run test, attach evals, and run the simulation.
+
+ How scenarios turn into scored calls
diff --git a/src/pages/docs/simulation/concepts/understanding-simulation.mdx b/src/pages/docs/simulation/concepts/understanding-simulation.mdx
new file mode 100644
index 00000000..0ca240f3
--- /dev/null
+++ b/src/pages/docs/simulation/concepts/understanding-simulation.mdx
@@ -0,0 +1,63 @@
+---
+title: "Understanding Simulation"
+description: "What goes into a simulation, and how the test-and-fix loop runs"
+---
+
+
+## What a simulation is
+
+A **simulation** runs your agent against simulated users so you catch its failures in a test instead of in production. The simulated user plays out a situation, your agent responds, and the whole conversation is scored by [evals](/docs/evaluation). It works the same way for a chat agent and for a voice agent on the phone. Run it before you ship, and again after every change, and you have a repeatable read on whether the agent is getting better or worse.
+
+## How a simulation works
+
+Three things go into a simulation:
+
+- the **[agent definition](/docs/simulation/concepts/agent-definitions)**, which is who gets tested and how Simulation reaches it: a phone number for a voice agent, your own code answering through the [SDK](/docs/simulation/reference/sdk-api) for a chat agent
+- the **[scenario](/docs/simulation/concepts/scenarios)**, the situation the conversation has to handle
+- the **[persona](/docs/simulation/concepts/personas)**, the character the simulated customer plays, attached to the scenario when you build it
+
+Future AGI's own agent, the **simulator**, plays the customer described by the persona and works through the scenario, turn by turn, against your agent. You don't configure it directly; you configure the three pieces it uses. When you set up the run you also pick the evals that score each conversation.
+
+
+
+You keep a library of personas and use them to build a range of scenarios. Each run loads one scenario into the simulated environment, where the simulator plays it out against your agent, and hands back a transcript, the metrics, and a score per eval. Build that library once and reuse it, so when a score moves it's the agent that changed, not the test. Each piece has its own page; everything else is how you run them and [read what comes back](/docs/simulation/guides/explore-results).
+
+## The test-and-fix loop
+
+Say you're putting a chat support agent in front of customers. In production it meets situations it never saw in development, and any one of them can go wrong in front of a customer. The loop looks like this:
+
+1. You write a refund request scenario and pick a frustrated caller persona
+2. You run it, and the resolution eval fails: the agent quotes the wrong refund window
+3. You shorten the agent's prompt and add the missing policy
+4. You re-run the same scenario with the same persona, and it passes
+
+Same test, changed agent, so once the score flips it's proof the fix worked, not a hunch. To run this loop on your own agent, start with [Run a chat simulation](/docs/simulation/guides/run-chat-simulation) or [Run a voice simulation](/docs/simulation/guides/run-voice-simulation).
+
+## Where Simulation fits
+
+Two neighbouring products meet Simulation:
+
+- **[Observe](/docs/observe)** watches real traffic; Simulation rehearses it before you ship
+- **[Evaluation](/docs/evaluation)** supplies the scoring: the same templates run in both, so a passing score means the same thing in a test and in production
+
+And two capabilities inside Simulation connect them:
+
+- **[Replay](/docs/simulation/concepts/replay)** turns a real conversation from Observe into a scenario you can rerun
+- **[Optimization](/docs/simulation/concepts/optimization)** picks up when a run exposes a weakness and improves the agent automatically
+
+## Keep exploring
+
+
+
+ The agent under test, and how versions track changes
+
+
+ The test cases that decide what happens
+
+
+ The simulated customer your agent faces
+
+
+ Put the loop to work on your own agent
+
+
diff --git a/src/pages/docs/simulation/features/evaluate-tool-calling.mdx b/src/pages/docs/simulation/features/evaluate-tool-calling.mdx
deleted file mode 100644
index 18e5529e..00000000
--- a/src/pages/docs/simulation/features/evaluate-tool-calling.mdx
+++ /dev/null
@@ -1,70 +0,0 @@
----
-title: "Evaluate Tool Calling: Agent Function Use in Simulations"
-description: "Evaluate the tool-calling capabilities of your AI agent during Future AGI simulation runs. Test whether agents invoke the correct tools and parameters."
----
-
-## About
-
-**Tool call evaluation** scores how well your agent uses tools during simulated conversations — checking whether it called the right tool, with the right arguments, at the right time. Enable it on a run test and after each conversation completes, the platform extracts every tool call and shows a Pass/Fail result with a reason alongside your other eval metrics.
-
-
-Your agent must be deployed with **tool calling enabled** to be evaluated. Enable tool call evaluation only for run tests where the agent under test actually uses tools.
-
-
----
-
-## When to use
-
-- **Check tool usage** — Confirm the agent invokes the right tools (e.g. transfer, end call) when it should and with the right inputs.
-- **Catch misuse** — See which tool calls failed evaluation (wrong tool, wrong arguments, or used at the wrong time).
-- **Compare across runs** — After changing prompts or tool definitions, re-run and compare tool-eval results to spot regressions.
-
----
-
-## How to
-
-You enable tool call evaluation when creating or editing a **run test** (agent-based simulation). It’s a toggle in the **Select Evaluations** step; for **voice** agents the platform may prompt you to provide **API Key** and **Assistant ID** so it can access your provider’s call data and extract tool calls.
-
-
-
- Go to **Simulate** → **Run Simulation**. Click **Create a Simulation** (or open an existing run). In **Add simulation details**, enter a name and select **Agent definition** and **Agent version** (for voice + tool eval, a version is required so the platform can use its API Key and Assistant ID). In **Choose Scenario(s)**, select one or more scenarios to run against. Click **Next** to go to Select Evaluations.
-
-
- In **Select Evaluations**, turn on **Enable tool call evaluation**. The platform will then run tool-call evaluation after each conversation in the run. You can also add other evaluations (task completion, tone, etc.) in this step. Click **Next** to go to Summary.
- 
-
-
- For **voice** agents, the platform needs your provider **API Key** and **Assistant ID** (for the agent you're evaluating) to fetch call data and extract tool calls. If the selected agent version doesn’t have these set, you’ll be prompted (e.g. **Update Keys for test**) when you enable tool call evaluation or when you save. Enter the API key and assistant ID; they are stored on the **agent version**. For **chat** agents, tool calls come from conversation data, so no keys are required.
- 
-
-
- In **Summary**, review the run test configuration, then create or save the run test. Open it and click **Execute** to start a test execution. When each conversation completes, the platform runs your evals and, if tool call evaluation is on, evaluates each tool call and attaches results to that call.
-
-
- Open the **execution detail** for the run. Tool call results appear with your other evaluation metrics—as columns or rows per tool call (e.g. “Transfer #1”, “End call #1”) with a result (Pass/Fail) and reason. Use them to see which tool calls passed or failed and why.
-
-
-
----
-
-## Notes
-
-- **Voice only:** API Key and Assistant ID are required for **voice** run tests when tool call evaluation is enabled, so the platform can pull call data from your provider. For chat run tests, tool calls are taken from stored conversation data.
-- **Agent version:** The keys are stored with the **agent version** you selected for the run test. If you switch to another version, you may need to update keys for that version if it uses a different assistant or provider.
-- **No tool calls:** If a conversation has no tool calls, nothing is evaluated for that call; other evals still run as usual.
-
----
-
-## Next Steps
-
-
-
- Create and execute simulation runs with tool call evaluation enabled.
-
-
- Define scenarios that trigger tool use so you can evaluate it.
-
-
- Configure your agent and versions (including tool-calling and provider credentials).
-
-
diff --git a/src/pages/docs/simulation/features/fix-my-agent.mdx b/src/pages/docs/simulation/features/fix-my-agent.mdx
deleted file mode 100644
index 835fd1aa..00000000
--- a/src/pages/docs/simulation/features/fix-my-agent.mdx
+++ /dev/null
@@ -1,296 +0,0 @@
----
-title: "Fix My Agent: Diagnostics and Fixes from Simulation Results"
-description: "Diagnose and fix agent performance issues using in-depth analytics from Future AGI simulation results. Get targeted recommendations for each failure type."
----
-
-
-
-After running simulations, Future AGI's **Fix My Agent** feature automatically analyzes your agent's performance and provides actionable recommendations to improve quality, reduce failures, and enhance overall effectiveness. Instead of manually debugging issues, get intelligent suggestions with one click.
-
----
-
-## About
-
-**Fix My Agent** analyzes your simulation results — call metrics, transcripts, and eval scores — and surfaces a prioritized list of issues with specific recommended fixes. After a run, instead of manually reviewing each call to find patterns, you get a clear breakdown of what's failing, how many calls it affected, and what to change. You can then implement fixes, re-run, and compare results to validate improvements.
-
-
-**Fix My Agent** gives you instant diagnostics and suggestions. For advanced prompt refinement, the platform also offers **optimization algorithms** (later in this guide) that automatically generate and test multiple prompt variations.
-
-
-## When to use
-
-- **Quick diagnostics** — Get instant, prioritized suggestions after every simulation run without manual debugging.
-- **Reduce failures** — Address high-priority issues (e.g. latency, brevity, end-of-speech) that affect the most calls.
-- **Validate changes** — Implement fixes, re-run the simulation, and compare metrics to confirm improvements.
-- **Auto-optimization (optional)** — Use algorithms (Random Search, Bayesian, Meta-Prompt, ProTeGi, PromptWizard, GEPA) to generate and evaluate optimized prompts when manual fixes aren’t enough.
-
-## How to
-
-Use **Fix My Agent** from the execution detail page after a simulation run. Recommended flow: run simulation → open Fix My Agent → review and apply suggestions → re-run to validate. Optionally run auto-optimization for systematic prompt refinement.
-
-
-
- After your simulation run completes, open the **execution detail** page.
-
- **What you see (field meanings):**
- | Field | Meaning |
- |-------|---------|
- | **Call Details** | Total calls, connected calls, connection rate for this run. |
- | **System Metrics** | CSAT scores, agent latency, WPM (words per minute). |
- | **Evaluation Metrics** | Results from the evaluations you attached to the simulation. |
-
- This is where **Fix My Agent** runs its analysis.
-
-
-
- Click **Fix My Agent** in the top-right of the execution page. A side panel opens.
-
- **What the panel shows (field meanings):**
- | Field | Meaning |
- |-------|---------|
- | **Suggestions** | Total number of issues the analysis identified. |
- | **Priority** | High / Medium / Low — urgency of each issue. |
- | **Issue categories** | Type of problem (e.g. latency, response brevity, detection tuning). |
- | **Affected calls** | How many calls in this run showed each issue. |
- | **Last updated** | When the analysis was last run (refresh to get a new analysis). |
-
- No configuration required—suggestions are generated from the run.
-
-
-
- Each suggestion in the panel has these parts:
-
- | Field | Meaning |
- |-------|---------|
- | **Issue description** | What's wrong (e.g. pipeline latency, response length, end-of-speech detection). |
- | **Recommended fix** | What to change (e.g. switch to a faster model, add a token limit, adjust VAD parameters). |
- | **Priority** | High / Medium / Low — tackle High first. |
- | **Affected calls** | Number of calls that showed this issue. |
- | **View issue** | Opens specific call examples so you can see the problem in context. |
-
- **Example suggestion types:** *Aggressively Reduce Pipeline Latency* (e.g. faster model for lower TTFT), *Enforce Strict Response Brevity* (e.g. hard token limit), *Tune End-of-Speech Detection* (e.g. adjust VAD). Implement the recommended changes in your system prompt, then re-run the simulation to validate. Start with High Priority; do 1–2 fixes per iteration and re-run to verify before moving on.
-
-
- To have the platform generate and test prompt variations, click **Optimize My Agent** in the Fix My Agent panel.
-
- **Configuration fields:**
- | Field | Meaning |
- |-------|---------|
- | **Name** | Label for this optimization run (e.g. "opt1", "latency-v2"). |
- | **Optimizer** | Algorithm that generates and evaluates prompt variations (see below). |
- | **Language model** | LLM used for the optimization (teacher model). |
- | **Parameters** | Optimizer-specific settings (e.g. number of variations, rounds, trials). |
-
- **Choose an optimizer** — Select from the algorithms below:
-
-
-
- **Best for:** Quick baseline testing and initial exploration.
-
- **How it works:** Generates random prompt variations using a teacher model and evaluates each candidate.
-
- **Characteristics:**
- - ⚡⚡⚡ Fast execution
- - ⭐⭐ Basic quality improvements
- - 💰 Low cost
- - Ideal for: 10-30 examples
-
- **Use when:** You need quick results or want to establish a performance baseline before trying more sophisticated algorithms.
-
-
- **Best for:** Few-shot learning tasks and intelligent example selection.
-
- **How it works:** Uses Bayesian optimization to intelligently select few-shot examples and prompt configurations.
-
- **Characteristics:**
- - ⚡⚡ Medium speed
- - ⭐⭐⭐⭐ High quality
- - 💰💰 Medium cost
- - Ideal for: 15-50 examples
-
- **Use when:** Your dataset contains good examples and you want to leverage few-shot learning effectively.
-
-
- **Best for:** Complex reasoning tasks requiring deep analysis.
-
- **How it works:** Analyzes failed examples, formulates hypotheses, and rewrites the entire prompt through deep reasoning.
-
- **Characteristics:**
- - ⚡⚡ Medium speed
- - ⭐⭐⭐⭐ High quality
- - 💰💰💰 Higher cost
- - Ideal for: 20-40 examples
-
- **Use when:** Your agent handles complex reasoning tasks or you need holistic prompt redesign.
-
-
- **Best for:** Identifying and fixing specific error patterns.
-
- **How it works:** Generates critiques of failures and applies targeted improvements using beam search to maintain multiple candidates.
-
- **Characteristics:**
- - ⚡ Slower execution
- - ⭐⭐⭐⭐ High quality
- - 💰💰💰 Higher cost
- - Ideal for: 20-50 examples
-
- **Use when:** You have clear failure patterns and want systematic error fixing.
-
-
- **Best for:** Creative exploration and diverse prompt variations.
-
- **How it works:** Combines mutation with different "thinking styles", then critiques and refines top performers.
-
- **Characteristics:**
- - ⚡ Slower execution
- - ⭐⭐⭐⭐ High quality
- - 💰💰💰 Higher cost
- - Ideal for: 15-40 examples
-
- **Use when:** You want creative exploration or diverse conversational approaches.
-
-
- **Best for:** Production deployments requiring state-of-the-art performance.
-
- **How it works:** Uses evolutionary algorithms with reflective learning and mutation strategies inspired by natural selection.
-
- **Characteristics:**
- - ⚡ Slower execution
- - ⭐⭐⭐⭐⭐ Excellent quality
- - 💰💰💰💰 Highest cost
- - Ideal for: 30-100 examples
-
- **Use when:** You need production-grade optimization with robust results and have sufficient evaluation budget.
-
-
-
- Click **Start Optimizing your agent** to begin the automated prompt generation process. The optimization engine will: (1) **Analyze** your simulation data and Fix My Agent suggestions; (2) **Generate** multiple system prompt variations using the selected algorithm; (3) **Evaluate** each variation against your test scenarios; (4) **Score** performance improvements; (5) **Select** the best-performing optimized prompt. View results in the **Optimization Runs** tab: performance comparison, best prompt, and history. Review the improved prompt, test on scenarios not in the original set, then update your agent and re-run to validate.
-
-
- Most users find that manually implementing **Fix My Agent** suggestions is the fastest path to improvement. Use auto-optimization when you need to test many prompt variations or want production-grade automated refinement.
-
-
-
- After implementing fixes or running auto-optimization, use the tabs below to view results and deploy.
-
-
-
- After implementing **Fix My Agent** suggestions:
-
- 1. **Re-run simulations** with your updated prompt
- 2. **Compare metrics** to baseline in the execution dashboard
- 3. **Review new suggestions** from Fix My Agent
- 4. **Iterate** until performance meets your goals
- 5. **Deploy** to production when satisfied
-
-
- If you used automated optimization, view results in the **Optimization Runs** tab:
-
- **Performance comparison** — Original prompt baseline scores, auto-generated prompt scores, improvement percentage.
-
- **Best prompt** — The highest-performing variation, changes from the original, evaluation scores across metrics.
-
- **Optimization history** — All variations tested, performance trajectory, iteration details.
-
- Copy the best prompt into your agent, test on new scenarios, then deploy. Always validate with test cases that weren't in the optimization set to avoid overfitting.
-
-
- Whether implementing manually or using auto-optimization:
-
- ✓ **Review** the improved prompt carefully
- ✓ **Test** with additional scenarios not in original dataset
- ✓ **Update** your agent definition with the new prompt
- ✓ **Re-run** simulations to validate improvements
- ✓ **Monitor** performance in production
-
-
- Always validate with new test cases before production deployment. Both manual and automated approaches can overfit to the evaluation dataset.
-
-
-
-
-
-
----
-
-### Algorithm Comparison
-
-| Algorithm | Speed | Quality | Cost | Best Dataset Size |
-|-----------|-------|---------|------|-------------------|
-| **Random Search** | ⚡⚡⚡ | ⭐⭐ | 💰 | 10-30 examples |
-| **Bayesian Search** | ⚡⚡ | ⭐⭐⭐⭐ | 💰💰 | 15-50 examples |
-| **Meta-Prompt** | ⚡⚡ | ⭐⭐⭐⭐ | 💰💰💰 | 20-40 examples |
-| **ProTeGi** | ⚡ | ⭐⭐⭐⭐ | 💰💰💰 | 20-50 examples |
-| **PromptWizard** | ⚡ | ⭐⭐⭐⭐ | 💰💰💰 | 15-40 examples |
-| **GEPA** | ⚡ | ⭐⭐⭐⭐⭐ | 💰💰💰💰 | 30-100 examples |
-
-
-- Speed: ⚡ = Slow, ⚡⚡ = Medium, ⚡⚡⚡ = Fast
-- Quality: ⭐ = Basic, ⭐⭐⭐⭐⭐ = Excellent
-- Cost: 💰 = Low, 💰💰💰💰 = High (based on API calls)
-
-
-### Decision Tree
-
-```
-Do you need production-grade optimization?
-├─ Yes → Use GEPA
-└─ No
- │
- Do you have clear error patterns to fix?
- ├─ Yes → Use ProTeGi
- └─ No
- │
- Is your task reasoning-heavy or complex?
- ├─ Yes → Use Meta-Prompt
- └─ No
- │
- Do you need few-shot learning optimization?
- ├─ Yes → Use Bayesian Search
- └─ No
- │
- Do you want creative exploration?
- ├─ Yes → Use PromptWizard
- └─ No → Use Random Search (baseline)
-```
-
----
-
-## Next Steps
-
-
-
- Learn how to run comprehensive agent simulations
-
-
-
- Build diverse test scenarios for better diagnostics
-
-
-
- Configure your agent for optimal performance
-
-
-
- Deep dive into auto-optimization algorithm details
-
-
-
----
-
diff --git a/src/pages/docs/simulation/features/observe-to-simulate.mdx b/src/pages/docs/simulation/features/observe-to-simulate.mdx
deleted file mode 100644
index 48147c44..00000000
--- a/src/pages/docs/simulation/features/observe-to-simulate.mdx
+++ /dev/null
@@ -1,95 +0,0 @@
----
-title: "Observe to Simulate: Replay Production Chat Sessions"
-description: "Replay real production sessions in a dev environment using chat simulation to debug, iterate, and improve your agent. Works with Observe data."
----
-
-## About
-
-**Replay** lets you take real production conversations captured in [Observe](/docs/observe) and rerun them against your dev agent using chat simulation. When something goes wrong in production, you select the exact session or trace, create a replay session, and run the same conversation end-to-end. Change your agent and replay again to verify fixes.
-
-### Replay types: session vs trace
-
-| Type | What is replayed | Use when |
-|------|------------------|----------|
-| **Session** | All traces in a given `session_id`, ordered by span start time — one multi-turn conversation per session. | You want to replay full production conversations as multi-turn chat scenarios. |
-| **Trace** | Each selected trace as a separate conversation with one turn (input → output). | You want to replay individual calls or single-turn interactions. |
-
-
-Replay does **not** require a new integration. It builds on **Observe** (to capture production sessions/traces) and **Chat Simulation** (to run the replayed conversations).
-
-
----
-
-## When to use
-
-- **Debug real failures**: Reproduce and fix issues from production instead of relying only on synthetic test cases.
-- **Reproduce edge cases**: Re-run conversations that only happened in production so you can iterate on them safely.
-- **Compare before vs after**: Change your agent and replay the same session to see how behavior and metrics change.
-- **Test fixes safely**: Validate prompt, model, or tool changes without impacting live users.
-- **Turn failures into regression tests**: Save the replayed scenario and add it to regular simulation runs.
-
----
-
-## How to
-
-You need **Observe** integrated (so production sessions and traces are in the platform), and **FI_API_KEY** / **FI_SECRET_KEY** for the replay and simulation APIs. To run the simulation via the SDK you’ll also need a **chat agent callback** and any LLM provider keys it uses — see [Chat Simulation Using SDK](/docs/simulation/features/simulation-using-sdk).
-
-The flow is: **select production data** → **create a replay session** → **generate scenario** (agent + scenario from transcripts) → **create run test** → **run simulation** → **view results and iterate**.
-
-
-
- With **Observe** integrated, your production system sends sessions and traces to the platform; they are stored per project. Once that data is there, you can create a replay session from it — no extra setup for replay.
-
-
- From the **Observe** experience (e.g. your project’s sessions or traces), choose what to replay: either **sessions** (full multi-turn conversations by `session_id`) or **traces** (individual traces, each treated as one turn). Create a **replay session** with:
-
- - **project_id** — The Observe project that owns the data.
- - **replay_type** — `"session"` or `"trace"`.
- - **ids** — List of session IDs or trace IDs to replay, **or** set **select_all** to include all sessions or all traces for the project.
-
- The platform creates a replay session in **INIT** and returns **suggestions** (e.g. `agent_name`, `scenario_name`, `agent_description`) and, if you already have replay sessions for this project, an existing **agent definition** to reuse. You can use these when generating the scenario in the next step.
-
-
- On the replay session, trigger **Generate scenario**. You provide:
-
- - **agent_name**, **scenario_name** (required); **agent_description** (optional).
- - **agent_type** — `"text"` (chat) or `"voice"`; for replay → chat simulation use **text**.
- - **no_of_rows** — How many scenario rows to generate from the transcripts (default 20).
- - Optional: **personas**, **custom_columns**, **graph**, **generate_graph**.
-
- The platform **creates or updates** an **agent definition** for the project, **creates a graph scenario** (source **Session Replay**) from the production transcripts, and starts the **scenario generation workflow**. The replay session moves to **GENERATING**. When the workflow finishes, the scenario is ready to use in a run test.
-
-
- Once the scenario is ready, **create a run test** that uses the replay session’s **agent definition** and **scenario**. When creating the run test, pass **replay_session_id** so the platform can mark the replay session as **COMPLETED** and link it to the new run test.
-
- Then **run the simulation** the same way you run any chat simulation: from the UI (**Simulate → Run Simulation**, then run the new run test) or via the **[Chat Simulation SDK](/docs/simulation/features/simulation-using-sdk)** (use the run test name and your agent callback). The replayed conversations run against your dev agent; transcripts and evals are stored in the dashboard.
-
-
- Open the **run test** (or simulation) and inspect the **test execution** and **call executions**. You get the same kind of results as for any chat simulation.
-
- **Performance metrics** (top of the execution view): **Chat details** — total chats, completed count, completion percentage. **System metrics** — avg output tokens, avg chat latency (ms), avg turn count, avg CSAT. **Evaluation metrics** — aggregated eval scores (e.g. ground truth match, task completion) showing how closely the replayed agent matches or improves on the original production behavior.
-
- **Session list** — Each row is one replayed session. Compare CSAT, token usage (total, input, output), and per-eval scores across runs. **Single session** — Click a session to see the **turn-by-turn transcript** (and, where available, a diff or comparison to the original production conversation) so you can see exactly where the agent’s responses, tool calls, or decisions changed after your fix.
-
- Update your agent (prompt, logic, tools, or model) and **replay again** to verify improvements.
-
-
-
----
-
-## Next Steps
-
-
-
- Run replayed (and other) simulations programmatically from your environment.
-
-
- Create and manage simulation runs and view executions.
-
-
- Understand scenarios and how replay creates graph scenarios from transcripts.
-
-
- Configure the agent used for replay and simulation.
-
-
diff --git a/src/pages/docs/simulation/features/prompt-simulation.mdx b/src/pages/docs/simulation/features/prompt-simulation.mdx
deleted file mode 100644
index a602dd37..00000000
--- a/src/pages/docs/simulation/features/prompt-simulation.mdx
+++ /dev/null
@@ -1,168 +0,0 @@
----
-title: "Prompt Simulation: Test Prompts in Multi-Turn Conversations"
-description: "Test your prompts in realistic multi-turn conversations directly from the Prompt Workbench, with no agent deployment or SDK required."
----
-
-## About
-
-**Prompt Simulation** lets you run your prompt template against realistic customer scenarios in multi-turn chat conversations — all from within the **Prompt Workbench**. Instead of waiting until after deployment to discover how your prompt performs in real conversations, you can test, evaluate, and iterate right away.
-
-When you run a simulation, the platform uses your **prompt version** as the "agent" and pairs it against a **simulated customer** driven by a scenario you define. Each scenario row becomes one chat conversation (up to 10 turns). When the conversations finish, any attached **evaluations** run automatically and produce scores and summaries you can act on immediately.
-
-
-Prompt simulation is distinct from agent-based simulation. You don't need an agent definition, an external deployment (e.g. Vapi, Retell), or any SDK code. Everything runs inside the Prompt Workbench.
-
-
----
-
-## When to use
-
-- **Test before you ship** — Run your prompt against realistic customer scenarios (refunds, support, onboarding) and review transcripts and eval scores before deploying to production.
-- **Compare prompt versions** — Create simulations for different saved versions of the same template and run them on the same scenarios to see which version performs better.
-- **Validate multi-turn behaviour** — See how your prompt handles follow-up questions, objections, or edge cases over several turns instead of judging it from single prompts in the Playground.
-- **Catch regressions** — After changing your prompt, re-run the same simulation and compare results so you spot unintended changes in tone, task completion, or safety.
-- **Tune evals** — Attach evaluations (task completion, tone, custom metrics) and use simulation runs to calibrate or improve your eval setup before using it on production traffic.
-- **No agent or SDK** — Get conversation-level feedback without building an agent definition or writing integration code; everything stays in the Prompt Workbench.
----
-
-## Key Concepts
-
-| Concept | What it is |
-|---|---|
-| **Prompt Template** | The container for your prompt (name, description, variable names). Lives in the Prompt Workbench. |
-| **Prompt Version** | A saved snapshot of the template (system message, model, parameters). The simulation uses one version as the "agent." |
-| **Scenario** | Defines who the simulated customer is and what they do. Types: `dataset`, `script`, or `graph`. Each row in a scenario → one chat session. |
-| **Persona** | Demographics and personality traits attached to a scenario. Controls how the simulated customer behaves (e.g. "frustrated buyer," "detail-oriented user"). |
-| **Simulation (Run Test)** | The saved config: which prompt version + which scenarios + which evals. Created from the Simulation tab. |
-| **Test Execution** | One run of a simulation. Created when you click Run Simulation. Tracks overall status and aggregated results. |
-| **Call Execution** | One chat session (one scenario row). Stores the transcript, eval outputs, token counts, and latency. |
-| **Eval Config** | An evaluation attached to the simulation. Runs automatically after each chat completes. |
-
----
-
-## How to
-
-Before you start: have a **prompt template** with at least one saved **prompt version** and at least one **scenario** (see [Scenarios](/docs/simulation/concepts/scenarios)).
-
-
-
- 1. Go to **Prompts** in the sidebar.
- 2. Open your prompt template.
- 3. Click the **Simulation** tab at the top of the workbench (next to Playground, Evaluation, and Metrics).
-
- 
- You'll see a list of existing simulations for this template, a **View Docs** button, and a **+ Create a Simulation** button. Click **+ Create a Simulation** to begin.
-
-
- Click **+ Create a Simulation**. The form walks you through four steps — complete each one and click **Next**; use **Back** to change earlier steps. **Next** stays disabled until required fields on the current step are filled.
-
-
-
- - **Simulation name** (required) — Enter a name for your simulation run (e.g. "Sales agent performance test" or "Refund flow - v3"). This identifies the simulation in the list.
- - **Choose Prompt version** (required) — Select the saved version of this prompt template that will act as the "agent" in every chat. The dropdown shows versions available for the current template.
- - **Description** (optional) — Describe what this simulation will evaluate (e.g. "Testing refund handling after prompt update").
- - Click **Next** to go to scenario selection.
- 
-
-
- - The screen says: **Choose your scenarios** — scenarios that your prompt will be tested against.
- - Use the **Search scenarios...** bar to find scenarios by name if you have many.
- - A list of scenarios is shown. Each row has: a **checkbox** to select, **Name** and **description**, a **type** tag (e.g. **Dataset**, **Graph**), and a **row count** (each row becomes one chat session when you run).
- - Select **at least one scenario**. You can select multiple; the total number of chats in a run is the sum of rows across selected scenarios.
- - Click **Next**. **Next** stays disabled until at least one scenario is selected.
- 
-
-
- - The screen says: **Select evaluations** — apply evaluation metrics to measure your prompt's performance.
- - **Enable tool call evaluation** — A toggle. When on, tool/function calls during chats will be evaluated. Turn it on only if your prompt uses tools.
- - **+ Add Evaluations** — Click to open the **Evaluations** picker. You can choose from pre-built evals (filter by Use Cases, Eval Categories, Eval Type; search by name) or create your own evals. Added evals appear in the list. Evals are optional. Click **Next** when done.
- 
-
-
- - **Review your simulation configuration before creating it.** Three sections: **Test Configuration** (name, prompt version), **Selected Test Scenarios** (count and details), **Selected Evaluations** (count). Click **Back** to fix anything; when satisfied, complete the flow to **create** the simulation. You're then taken to the **simulation detail** view.
- - If you're asked to **Update Keys for test** (e.g. API Key, Assistant ID), fill in the required fields and save.
-
-
-
-
- From the simulation detail view you can adjust settings before running:
- - **Version** — Switch which prompt version is used. Useful for A/B comparisons between versions on the same scenario set.
- - **Scenarios** — Add or remove scenarios. At least one is required to run.
- - **Evals** — Add, edit, or remove evaluation configs. Evals run automatically after each chat completes.
-
-
- 1. On the simulation detail view, click **Run Simulation** in the top-right corner.
- 2. A confirmation notification appears and the run begins.
- 
-
- The platform will: create one **test execution** for this run; resolve all attached scenarios into rows; create one **call execution** (chat session) per row; run each chat (your prompt version as the agent, the scenario's simulator as the customer, up to **10 turns** per conversation); run all attached eval configs after each chat completes.
-
-
- You can run the same simulation multiple times (e.g. after changing your prompt version or scenarios). Each click of Run Simulation creates a new test execution, so all historical runs are preserved.
-
-
-
- Click any **execution row** on the simulation detail view to open **Execution Detail**. Here you see a run-level summary at the top and a list of every chat below; use the tabs to understand what each area shows and how to use it.
-
-
-
- The **top panel** gives you a quick read on the whole run. Use it to see overall health (how many chats completed), cost (tokens), and how your prompt scored on the evals you attached.
-
- | Metric group | What you see |
- |---|---|
- | **Chat Details** | Total chats, how many completed, and completion percentage — tells you whether the run finished cleanly or had failures. |
- | **System Metrics** | Average total, input, and output tokens per chat, and average latency (ms). Use this to spot high-cost or slow conversations. |
- | **Evaluation Metrics** | Average score for each evaluation you configured (e.g. Task Completion, Tone). Click **View all metrics** for a full breakdown across evals and chats. |
-
- Use these numbers to compare runs (e.g. before vs after a prompt change) or to spot runs that need a closer look in the grid.
-
-
- The **grid** lists every chat (one per scenario row). Each row is one conversation: status, scores, and usage. Use it to find failed or low-scoring chats, compare behaviour across scenarios, or pick chats to drill into.
-
- | Column | What it tells you |
- |---|---|
- | **Chat Details** | Status (Completed / Failed), start time, and number of turns. Use status to quickly find failures. |
- | **CSAT** | Customer satisfaction score for that chat, with a color indicator. |
- | **Total / Input / Output Tokens** | Token usage for that conversation — useful for cost and length. |
- | **Average Latency (ms)** | How long the model took to respond on average in that chat. |
- | **Turn Count** | Number of back-and-forth exchanges (up to 10 per run). |
- | **Evaluation Metrics** | Per-eval results as tags (e.g. Tone: Joy, Neutral, Annoyance). Scan to see which chats passed or failed which evals. |
-
- Use the **Search** bar and **Filter** icon to narrow by status, score, or other criteria.
-
-
- **Drill into one conversation:** click any **chat row** in the grid to open that chat's detail view. You get the **full transcript** (every message from your prompt and the simulated customer), plus that chat's **eval scores** and **token/latency breakdown**. Use this to see why a chat failed an eval, how the model responded to tricky turns, or to copy a conversation for debugging or training.
-
-
-
-
- | Action | How |
- |---|---|
- | **Re-run simulation** | Click **Re-run** from the execution detail to run the same simulation again. |
- | **Rerun selected calls** | Rerun only certain chats from an execution. |
- | **Rerun whole execution** | Rerun all chats in that execution. |
- | **Cancel a run** | Stop a run in progress. |
- | **Export data** | Download results as CSV. |
- | **Fix My Agent** | AI-powered suggestions to improve your prompt. |
- | **Add More Evals** | Attach more evaluations and run on completed conversations. |
-
-
-
----
-
-## Next Steps
-
-
-
- Learn how to create scenarios with datasets and personas.
-
-
- Use AI-powered suggestions to improve your prompt based on simulation results.
-
-
- Build evaluations tailored to your specific use case.
-
-
- Run simulations against a deployed voice or chat agent programmatically.
-
-
diff --git a/src/pages/docs/simulation/features/run-simulation.mdx b/src/pages/docs/simulation/features/run-simulation.mdx
deleted file mode 100644
index 66ce4f4a..00000000
--- a/src/pages/docs/simulation/features/run-simulation.mdx
+++ /dev/null
@@ -1,64 +0,0 @@
----
-title: "Run Voice Simulation: Test Agents Against Scenarios"
-description: "Create and run voice simulation tests from the Future AGI platform to evaluate your agent against predefined scenarios and personas."
----
-
-## About
-
-Running a simulation from the platform means creating a test that combines your agent definition, one or more scenarios, and evaluation configs. The platform runs the conversations (voice calls or chat), records transcripts and metrics, and scores every interaction with the evaluations you configure.
-
-Before running a simulation, you need:
-
-- An [Agent Definition](/docs/simulation/concepts/agent-definition) configured for your agent
-- One or more [Scenarios](/docs/simulation/concepts/scenarios)
-- Optionally, [Personas](/docs/simulation/concepts/personas) assigned to your scenarios
-
-## How to
-
-
-
-
-Go to **Simulate > Tests** in the sidebar. Click **Create Test**.
-{/* TODO: Add screenshot of test list page with Create Test button */}
-
-
-
-Fill in the test details.
-{/* TODO: Add screenshot of step 1 of the wizard */}
-{/* TODO: Verify the exact fields shown in the wizard and add a field reference table */}
-
-
-
-Search and select one or more scenarios. Each scenario generates one or more simulated conversations.
-{/* TODO: Add screenshot of scenario selection step */}
-
-
-
-Add evaluation configs that will score every conversation. Optionally enable tool call evaluation.
-{/* TODO: Add screenshot of evaluation selection step */}
-
-
-
-Review the summary of your test configuration. Click **Create** to start the test.
-{/* TODO: Add screenshot of summary step */}
-
-
-
-
-## After Running
-
-Once the test starts, you can monitor progress from the test detail page. See [View Results](/docs/simulation/features/view-results) for how to read scores, transcripts, and analytics.
-
-## Next Steps
-
-
-
- Read transcripts, evaluation scores, and performance analytics for your test runs.
-
-
- Validate that your agent calls the right tools with the right parameters.
-
-
- Use optimization runs to automatically improve your agent based on test results.
-
-
diff --git a/src/pages/docs/simulation/features/simulation-using-sdk.mdx b/src/pages/docs/simulation/features/simulation-using-sdk.mdx
deleted file mode 100644
index 354266e5..00000000
--- a/src/pages/docs/simulation/features/simulation-using-sdk.mdx
+++ /dev/null
@@ -1,151 +0,0 @@
----
-title: "Chat Simulation Using SDK: Run Tests from Python"
-description: "Run Future AGI chat simulations from Python by providing an agent callback and executing Run Tests. Automate and scale simulation testing with the SDK."
----
-
-## About
-
-**Chat simulation using the SDK** lets you run an existing chat simulation from your own code. The platform drives the customer side using your scenarios. For each turn, it sends the simulator message to your **agent callback**, your code returns the reply, and the SDK posts it back. This continues until the conversation ends or the turn limit is reached. Transcripts and evaluation results are stored in your dashboard.
-
-
-You need a chat simulation already created in the UI (**Simulate > Run Simulation**). The SDK runs it by **name** (exact match). Your agent lives in your code; the platform stores results under the same simulation.
-
-
----
-
-## When to use
-
-- **Run from code**: Execute chat simulations from Python or CI instead of the UI using your existing agent implementation.
-- **Test your own agent**: Plug in any chat agent (LangChain, LlamaIndex, custom) via a single callback. No need to deploy to the platform first.
-- **Same config as UI**: Same Run Test, scenarios, and evals as the UI. Only the “agent” is your callback.
-- **Automate and iterate**: Script simulations, run many configs, and inspect transcripts and evals in the dashboard.
-
----
-
-## How to
-
-You need: Python 3.10+, **FI_API_KEY** and **FI_SECRET_KEY**, a **chat simulation** created in the UI, and (if your callback uses an LLM) the relevant provider key (e.g. OPENAI_API_KEY). Create the simulation in the UI, then either use the SDK drawer to copy the code or follow the steps below; results appear in the dashboard under that simulation.
-
-
-
- Go to **Simulate → Run Simulation → Create a Simulation**. Use a **chat** agent definition and version, add scenarios and optional evals, then save. Open the simulation from **Simulate → Run Simulation** by clicking it — you’re on the simulation detail (e.g. **Simulated runs** tab). For full setup (agent, scenarios, personas), see [Run simulation](/docs/simulation/features/simulation-using-sdk).
-
-
- On the simulation detail page, click **Run New Simulation**. For **chat** agent simulations (non–prompt), the UI does not call the execute API; it opens a **right-side drawer** with SDK instructions: **Step 1** — install the SDK (copy/run the snippet); **Step 2** — create a simulation run (copy/run the code to start the simulation from your environment). You can use that code as-is or follow the steps below. The drawer content comes from the same install and run snippets described in this guide.
-
-
- ```bash
- pip install agent-simulate litellm
- ```
- `litellm` is optional; use it if you want to call OpenAI/Anthropic/Gemini from the example. Set **FI_API_KEY** and **FI_SECRET_KEY** (env vars or pass into `TestRunner`). If your callback calls an LLM, set the provider key (e.g. OPENAI_API_KEY).
-
-
- Your callback receives **AgentInput** (each turn the simulator sends `thread_id`, `messages`, `new_message`, `execution_id`) and returns a **string** or **AgentResponse** (`content`, and optionally `tool_calls`, `tool_responses`, `metadata`). You can use a plain async function or the **AgentWrapper** class — implement `async def call(self, input: AgentInput) -> Union[str, AgentResponse]` and pass an instance as `agent_callback`.
-
- - **input.new_message** — The latest simulator message you should respond to (the “user” message for this turn).
- - **input.messages** — Full conversation history so far (including the latest simulator message).
- - **input.thread_id** / **input.execution_id** — For logging or correlation.
-
- If your agent uses **tools**, return an **AgentResponse** with `content`, `tool_calls`, and (if you have them) `tool_responses`; you can mock tool outputs inside the callback.
-
-
-
- ```python
- async def agent_callback(input: AgentInput) -> Union[str, AgentResponse]:
- user_text = (input.new_message or {}).get("content", "") or ""
- # Call your LLM or logic; return str or AgentResponse
- return "Your reply"
- ```
-
-
- ```python
- from fi.simulate import AgentWrapper, AgentInput, AgentResponse
- from typing import Union
-
- class MyAgent(AgentWrapper):
- async def call(self, input: AgentInput) -> Union[str, AgentResponse]:
- user_text = (input.new_message or {}).get("content", "") or ""
- return f"You said: {user_text}"
-
- # await runner.run_test(..., agent_callback=MyAgent(), ...)
- ```
-
-
- Return `AgentResponse` with `content`, `tool_calls`, and `tool_responses`:
- ```python
- return AgentResponse(
- content="Let me look that up.",
- tool_calls=[{"id": "call_1", "type": "function", "function": {"name": "lookup_order", "arguments": '{"order_id": "123"}'}}],
- tool_responses=[{"role": "tool", "tool_call_id": "call_1", "content": '{"status": "shipped"}'}],
- )
- ```
-
-
-
-
- You can keep your existing chat agent (LangChain, LlamaIndex, custom app) and wrap it in `agent_callback` so the simulator gets replies turn-by-turn.
-
-
-
- Create a `TestRunner` with your API key and secret, then call `run_test` with the **exact simulation name** (the name shown in Simulate → Run Simulation) and your callback. Run this code in your terminal or script — the SDK talks to the backend and creates/runs the simulation; results then show under the same simulation (e.g. **Simulated runs** tab):
-
- ```python
- from fi.simulate import TestRunner, AgentInput, AgentResponse
- import litellm
- import os
- from typing import Union
- import asyncio
-
- FI_API_KEY = os.environ.get("FI_API_KEY", "")
- FI_SECRET_KEY = os.environ.get("FI_SECRET_KEY", "")
- run_test_name = "Chat test" # must match simulation name in UI (Simulate → Run Simulation)
- concurrency = 5
-
- async def agent_callback(input: AgentInput) -> Union[str, AgentResponse]:
- user_text = (input.new_message or {}).get("content", "") or ""
- resp = await litellm.acompletion(
- model="gpt-4o-mini",
- messages=[{"role": "user", "content": user_text}],
- temperature=0.2,
- )
- return resp.choices[0].message.content or ""
-
- async def main():
- runner = TestRunner(api_key=FI_API_KEY, secret_key=FI_SECRET_KEY)
- await runner.run_test(
- run_test_name=run_test_name,
- agent_callback=agent_callback,
- concurrency=concurrency,
- )
- print("Simulation completed. View results in the dashboard.")
-
- asyncio.run(main())
- ```
-
-
- You can run the full notebook in Colab: [Chat Simulate Testing.ipynb](https://colab.research.google.com/drive/167WDQHSUZbuQ9GrszNUWK6etLm6D8M2o?usp=sharing).
-
-
-
- Transcripts, metrics, and evaluations appear under the **same simulation** in the dashboard (e.g. **Simulated runs** tab). Open the simulation → select the test execution → open a call execution to see the full transcript and eval results. The SDK orchestrates runs and supplies agent replies; the platform stores all results.
-
-
-
----
-
-## Next Steps
-
-
-
- Create and manage simulations in the UI (Simulate → Run Simulation).
-
-
- Build chat scenarios for your simulation.
-
-
- Configure your chat agent in the UI.
-
-
- Test prompts in multi-turn chat from the Workbench (no SDK).
-
-
diff --git a/src/pages/docs/simulation/features/view-results.mdx b/src/pages/docs/simulation/features/view-results.mdx
deleted file mode 100644
index 15b5656d..00000000
--- a/src/pages/docs/simulation/features/view-results.mdx
+++ /dev/null
@@ -1,121 +0,0 @@
----
-title: "View Simulation Results: Transcripts, Scores, and Analytics"
-description: "Read simulation results in Future AGI: view conversation transcripts, evaluation scores, performance analytics, and call logs for each agent run."
----
-
-## About
-
-After a simulation test runs, the results are available in the test detail view. You can see every conversation transcript, evaluation scores per call, aggregated analytics across all runs, and detailed per-call metrics including latency, token usage, and cost.
-
----
-
-## Test Detail View
-
-When you open a completed test, you see three tabs:
-
-### Simulated Runs
-
-A list of all test executions. Each execution represents one run of the test. Click an execution to see its individual call results.
-
-{/* TODO: Add screenshot of simulated runs tab */}
-
-### Logs
-
-Every call execution for this agent. Each row shows call information (duration, participants, status), and evaluation scores when the run had evals configured. You can filter by version to see only calls that used a specific agent version.
-
-
-
-### Analytics
-
-Performance analytics showing how your agent performed across test runs:
-
-
-
-- **Call success rate**: Proportion of calls that completed successfully vs failed or cancelled
-- **Average response time**: How long the agent typically takes to respond
-- **Evaluation scores**: Scores by eval metric (correctness, tone, compliance) so you can see which areas are strong or weak
-- **Error rate**: How often calls fail or hit errors
-
-Use this to track performance over time, compare across versions, and spot regressions before shipping.
-
----
-
-## Inspecting a Call
-
-
-
- Go to the **Logs** tab. Optionally filter by version using the version selector.
-
-
- Click a call in the list. A call detail view opens with the full conversation and results.
-
- 
-
-
- In the call detail you get:
-
- - **Full transcript**: Turn-by-turn conversation between agent and simulated customer
- - **Evaluation results**: Scores per metric for this specific call
- - **Audio playback**: When available for voice simulations
- - **Cost breakdown**: Token usage and cost for this call
- - **Trace information**: Detailed tracing data if observability is enabled (see [Observe](/docs/observe))
-
-
-
----
-
-## Execution Detail
-
-Click on a specific execution from the Simulated Runs tab to see detailed results.
-
-### Call/Chat Details
-
-The main view shows every conversation in this execution with transcripts, metadata, evaluation scores, and conversation flow visualization.
-
-{/* TODO: Add screenshot of execution detail call/chat view */}
-
-### Analytics
-
-Performance metrics for this specific execution:
-- Latency distribution
-- Token usage
-- System metrics
-- Evaluation score breakdown by metric
-
-{/* TODO: Add screenshot of execution analytics tab */}
-
-### Optimization Runs
-
-If you've run [Fix My Agent](/docs/simulation/features/fix-my-agent) or other optimizations on this execution, they appear here with their status and results.
-
----
-
-## Side Drawer
-
-Clicking on a specific call opens a detail drawer showing:
-- Scenario details for this call
-- Evaluation results grid with pass/fail per metric
-- Poor evaluations highlighted
-- System metrics (latency, tokens)
-- Cost breakdown
-- Call analytics summary
-- Trace information
-- Baseline comparison option (compare against a previous version's results)
-
-{/* TODO: Add screenshot of side drawer */}
-
----
-
-## Next Steps
-
-
-
- Get AI-powered diagnostics and optimization suggestions based on results.
-
-
- Create and run another test with different scenarios or agent versions.
-
-
- Score tool-calling performance during simulations.
-
-
diff --git a/src/pages/docs/simulation/features/voice-replay.mdx b/src/pages/docs/simulation/features/voice-replay.mdx
deleted file mode 100644
index 6b75e1d4..00000000
--- a/src/pages/docs/simulation/features/voice-replay.mdx
+++ /dev/null
@@ -1,115 +0,0 @@
----
-title: "Voice Replay: Debug Voice Agents from Production Calls"
-description: "Replay real production voice calls from Future AGI Observe in simulation to debug, iterate, and improve your voice agent based on real usage."
----
-
-## What it is
-
-**Voice Replay** (Observe → Simulate) lets you **replay real production voice calls** captured via **Voice Observability** and rerun them in a **development environment** using **voice simulation**. When something goes wrong in production -a misunderstood order, wrong tool call, poor latency, or bad tone -you can select the exact **voice trace** from Observe, create a **replay session**, turn it into a **simulation scenario**, and run a new voice call end-to-end against your dev agent. Change your agent (prompt, model, voice settings) and replay again to verify fixes. This closes the loop between **voice observability** and **iteration**.
-
-Under the hood, the platform extracts the **original voice configuration** (system prompt, assistant settings, provider config) from the production trace's raw call log, creates a **voice agent definition** with a configuration snapshot matching the original call, and generates a **graph scenario** from the production conversation. You then run the scenario via **Voice Simulation** (UI or SDK). Results include **side-by-side transcript comparison**, **performance metrics comparison**, and **audio recording playback** for both the baseline and replayed calls.
-
-
-Voice Replay currently supports **Vapi** as the primary provider. **Retell** is supported for transcript comparison but config extraction during replay setup is optimized for Vapi's data structure.
-
-
-***
-
-## Use cases
-
-- **Debug voice agent failures** -Reproduce misunderstood intents, wrong tool calls, or hallucinations from real production calls.
-- **Compare call quality** -Replay the same conversation after changing your prompt, model, or voice settings and compare latency, WPM, and talk ratio side by side.
-- **Test provider changes** -Switch from one voice provider or model to another and replay the same scenarios to measure impact.
-- **Iterate on voice UX** -Improve first messages, interruption handling, or response length by replaying real caller interactions.
-- **Turn failures into regression tests** -Save the replayed scenario and add it to regular simulation runs or CI.
-
-***
-
-## How to
-
-You need **Voice Observability** integrated (so production voice calls are captured with their recordings and transcripts), and **FI_API_KEY** / **FI_SECRET_KEY** for the replay and simulation APIs.
-
-The flow is: **select voice traces** → **create a replay session** → **generate scenario** (agent + scenario from audio/transcripts) → **create run test** → **run voice simulation** → **compare with baseline and iterate**.
-
-
-
- With **Voice Observability** integrated, your production voice calls (via Vapi, Retell, or other supported providers) are captured as traces with conversation-type spans. Each span stores the full call data including transcripts, recordings, and call metrics. See [Set Up Voice Observability](/docs/observe/concepts/voice-observability) for integration details.
-
-
- From the **Observe** experience, select the voice traces you want to replay. Create a **replay session** with:
-
- - **project_id** -The Observe project that owns the voice traces.
- - **replay_type** -`"trace"` (each voice trace is one complete call).
- - **ids** -List of trace IDs to replay, **or** set **select_all** to include all voice traces.
-
- The platform detects that these are voice traces (by checking for conversation-type spans), extracts the **original voice configuration** from the raw call log (system prompt, assistant ID, provider, model, phone number), and returns **suggestions** including `agent_type: "voice"` and the extracted config.
-
- 
-
-
- On the replay session, trigger **Generate scenario**. You provide:
-
- - **agent_name**, **scenario_name** (required); **agent_description** is auto-extracted from the original call's system prompt.
- - **agent_type** -`"voice"`.
- - **no_of_rows** -How many scenario rows to generate (default 20).
-
- 
-
- The platform:
- 1. **Creates a voice agent definition** with the original provider config (assistant ID, model, voice settings) preserved in the agent version's configuration snapshot.
- 2. **Extracts user intents** from each trace -if recording URLs are available, the audio is used for intent extraction. If no recordings exist, text transcripts are used as a fallback.
- 3. **Generates a graph scenario** (source **Session Replay**) with persona, situation, and outcome columns derived from the call data.
-
- The replay session moves to **GENERATING**. When the workflow finishes, the scenario is ready.
-
- 
-
- Once generated, you can review the scenario rows with persona, situation, and outcome details.
-
- 
-
-
- After scenarios are generated, you can optionally **map eval variables** -connect scenario columns (like expected outcome or situation context) to evaluation metrics so the platform can automatically score each replayed call. You can also add additional evaluations after the replay.
-
- Then click **Start Replay** to create a run test linked to the replay session.
-
-
- Once the run test is created, **run the voice simulation** -the platform calls the voice provider using the preserved configuration snapshot, so the replayed call uses the same assistant settings, model, and voice as the original production call. Each scenario row generates a new voice call.
-
-
- After the simulation completes, open a call execution and click **Compare with baseline call** to see a side-by-side comparison:
-
- **Performance metrics** -Call Duration, Turn Count, Avg Agent Latency (ms), User WPM, Bot WPM, and Talk Ratio, each showing the value, absolute change, and percentage change from the baseline call.
-
- **Audio recordings** -Play back both the baseline and replayed call recordings (stereo, mono combined, mono customer, mono assistant) directly in the UI.
-
- **Transcript comparison** -Side-by-side transcripts of the baseline call and the replayed call. Use **Show Diff** to highlight differences between the two conversations.
-
- Update your agent (prompt, model, voice settings, or tools) and **replay again** to verify improvements.
-
- 
-
-
-
-
-The **Compare with baseline call** button only appears for call executions that originated from a replay session (where a baseline trace exists to compare against).
-
-
-***
-
-## What you can do next
-
-
-
- Replay text-based production sessions using chat simulation.
-
-
- Set up voice call monitoring for production calls.
-
-
- Understand scenarios and how replay creates graph scenarios from transcripts.
-
-
- Configure voice agents for simulation, including provider settings and voice config.
-
-
diff --git a/src/pages/docs/simulation/guides/connect-your-agent.mdx b/src/pages/docs/simulation/guides/connect-your-agent.mdx
new file mode 100644
index 00000000..afede14e
--- /dev/null
+++ b/src/pages/docs/simulation/guides/connect-your-agent.mdx
@@ -0,0 +1,87 @@
+---
+title: "Connect your agent"
+description: "Create an agent definition for your voice or chat agent, and version it as you edit"
+---
+
+Simulation can only test an agent it can reach, and the [agent definition](/docs/simulation/concepts/agent-definitions) is where you tell it how. This guide walks the create wizard for a voice agent, covers where the chat path differs, and shows how to freeze the configuration you tested as a version.
+
+## Open Agent Definitions
+
+Go to **Agent Definition** under **Simulate** in the sidebar. The page lists every definition in your workspace with its type, provider, contact number, and current version. Click **Create agent definition** at the top right.
+
+
+*Everything starts from Create agent definition on the Agent Definitions page*
+
+## Name the agent and pick its type
+
+The **Basic Info** step asks who this agent is. The **Agent type**, voice or chat, decides how Simulation reaches the agent, so the rest of the wizard follows from it: a voice agent is dialed over the phone, a chat agent is answered by your own code. Give the agent a clear name and select the languages its conversations run in.
+
+
+*Pick the agent type and languages on Basic Info; this guide follows the voice path*
+
+## Choose the provider
+
+On the **Configuration** step, pick the **Voice/Chat Provider** powering your agent. **Vapi** and **Retell** are integrated natively, so choosing one lets the definition sync details straight from the provider in the next step. Choose **Others** for an agent on any other stack: a phone number Simulation can dial is all it needs.
+
+
+*Vapi, Retell and Bland.ai are native; Others covers any agent reachable by phone*
+
+## Add the connection details
+
+With **Vapi** selected, the provider fields appear. Fill them in top to bottom:
+
+- **Authentication Method**: choose **API Key**, then paste your provider API key
+- **Assistant ID**: the assistant to test; a successful sync pulls its name and system prompt from the provider, so the definition matches what runs in production
+- **Enable observability**, optional: turn it on to track the agent's calls and logs for debugging later
+- **Contact Information**: the country code and the phone number calls are routed to or from, with **Inbound Calls** left on if the agent takes incoming calls
+
+If the sync fails, the Assistant ID field flags it: recheck the API key and the ID, and the synced fields fill in on their own once both are right. **Retell** asks for the same details; with **Others** there are no provider credentials, just the contact number.
+
+
+*Provider credentials, the contact number, and the inbound toggle live on Configuration*
+
+## Set the behaviour and create
+
+The **Behaviour** step holds the agent's own instructions. **Prompt / Chains** carries the agent's system prompt; if you synced from Vapi or Retell it arrives prefilled with the provider's prompt, otherwise write it here. You can also attach a [knowledge base](/docs/knowledge-base) so the agent answers from your domain material. The **Commit Message** works the way it does in code: a short line describing this configuration, stored on the version it becomes. Check the summary on the right, then click **Create agent definition**.
+
+
+*The system prompt, knowledge base, and commit message, then Create agent definition*
+
+## Connecting a chat agent
+
+Pick **Chat** as the agent type on Basic Info and the wizard keeps the same three steps, but the connection changes shape: there's no provider to pick and no number to add, because a chat agent is answered by your own code. The Configuration step asks only which model your agent uses, and Basic Info and Behaviour work exactly as above.
+
+
+*Configuration on a chat agent is one field: the model it runs on*
+
+The connection itself happens when you run: with the `agent-simulate` SDK you attach your agent as a callback, each turn the [persona](/docs/simulation/concepts/personas) says arrives at that callback, and whatever it returns is your agent's reply, until the conversation ends. [Run a chat simulation](/docs/simulation/guides/run-chat-simulation) walks through the run itself, and the [SDK & API reference](/docs/simulation/reference/sdk-api) has the code your service starts from.
+
+## Version the definition as you edit
+
+The definition you just created is version 1, carrying the commit message you wrote in the wizard. Editing the definition later changes its live configuration and touches no version, so nothing you've already tested shifts under you.
+
+To freeze the current configuration, open the definition from the list and click **Create new version** in its **Version Management** panel. The drawer shows the configuration you're about to freeze, with room for final edits, and asks what's changing in this version, the same commit message the wizard asked for. That version becomes the active one, the default for runs where you don't pick a version at run setup; older versions stay runnable, which is how you re-run a configuration you've edited past.
+
+
+*Create new version freezes the configuration under the next number, here v3*
+
+## Dive deeper
+
+
+
+ Build the situations your agent gets tested on
+
+
+ Shape the customer on the other side of the call
+
+
+ Put the agent you just connected through a run
+
+
diff --git a/src/pages/docs/simulation/guides/create-personas.mdx b/src/pages/docs/simulation/guides/create-personas.mdx
new file mode 100644
index 00000000..d0bea426
--- /dev/null
+++ b/src/pages/docs/simulation/guides/create-personas.mdx
@@ -0,0 +1,95 @@
+---
+title: "Create personas"
+description: "Build a simulated customer persona to test your agent against"
+---
+
+A custom [persona](/docs/simulation/concepts/personas) lives in your workspace, and any [scenario](/docs/simulation/concepts/scenarios) can draw on it.
+
+
+Before building one, look through the **Future AGI Built** tab on the Personas page: 18 ready-made personas covering common customer types, all listed in the [Built-in personas](/docs/simulation/reference/built-in-personas) reference. Build your own only when none of them matches the customer you have in mind.
+
+
+## Creating custom personas
+
+Under **Simulate** in the sidebar, open **Personas** and click **Create persona**.
+
+
+*The Personas page is the pool your scenarios draw from*
+
+## Choose voice or chat
+
+Pick the channel. A persona is typed at creation and the type never changes: when you later build a voice simulation, only voice personas are offered, and the same goes for chat.
+
+
+*Voice or chat, fixed at creation*
+
+## Describe the customer
+
+**Basic Information** takes a name and a one-line description, both required. The description steers the simulator more than any single trait, so make it concrete: "a customer who is angry about the product" beats "difficult customer". Below them sit the optional demographics: gender, age range, location, and profession.
+
+**Behavioural Settings** shapes how that customer comes across: personality traits like impatient and direct or cautious and skeptical, a communication style, and on a voice persona the accent. Anything you leave unset falls back to a default.
+
+
+*Basic Information and Behavioural Settings are the two panels of the form*
+
+## Tune the conversation
+
+How the persona holds a conversation depends on its type.
+
+### Voice settings
+
+**Conversation Settings** on a voice persona controls the mechanics of the call:
+
+- **Multilingual and language**: the language the persona speaks, and whether it switches between several during the call
+- **Conversation speed**: how fast it talks, from 0.5x to 1.5x
+- **Background noise**: play real-world noise behind the customer, to test how your agent copes with an imperfect line
+- **Finished Speaking Sensitivity**, 1 to 10: how quickly it starts talking after your agent pauses; at 10 it jumps in after the shortest pause
+- **Interrupt Sensitivity**, 1 to 10: how easily it stops talking when your agent speaks over it; at 1 it doesn't respond to interruptions at all
+
+
+*The Conversation Settings panel of a voice persona*
+
+### Chat settings
+
+A chat persona swaps the call mechanics for writing style:
+
+- **Tone**: formal, neutral, or casual
+- **Verbosity**: brief, balanced, or detailed replies
+- **Regional Mix**: how much local phrasing colours the writing, from none to heavy
+- **Slang Level**: from none to heavy
+- **Typo Level**: how often typos slip into the messages, from none to frequent
+- **Punctuation Style**: clean, minimal, expressive, or erratic
+- **Emoji Frequency**: from never to heavy
+
+Every one has a default, so here too you only set what matters to the test.
+
+
+*The Chat Settings panel of a chat persona*
+
+## Custom properties and instructions
+
+**Add custom properties** takes key-value pairs you name yourself, like `insurance_type: renters`, and they travel with the persona into every scenario row generated from it. **Additional instructions** is free-form guidance the simulator always follows, like "ask for a supervisor after the first objection". Both are optional: when the built-in fields already cover your customer, skip straight to **Save**.
+
+
+*Custom properties and additional instructions on the create form*
+
+## Save and find it under Custom
+
+Click **Save**. The persona lands under the **Custom** tab of the Personas list, and from here it works exactly like a built-in one: you attach it when you [create a scenario](/docs/simulation/guides/create-scenarios), and the scenario carries it into every run. To adjust it later, open it again from the edit icon on its row; only your personas carry one, built-in personas can't be edited.
+
+
+*Your personas, ready for any scenario in the workspace*
+
+## Dive deeper
+
+
+
+ Attach your persona to the conversations it should play
+
+
+ All 18 ready-made personas, and every field they can set
+
+
+ Put the persona on a call with your agent
+
+
diff --git a/src/pages/docs/simulation/guides/create-scenarios.mdx b/src/pages/docs/simulation/guides/create-scenarios.mdx
new file mode 100644
index 00000000..8fb8db9b
--- /dev/null
+++ b/src/pages/docs/simulation/guides/create-scenarios.mdx
@@ -0,0 +1,114 @@
+---
+title: "Create scenarios"
+description: "Build the test cases a simulation runs, from a graph, a dataset, a script, or an SOP"
+---
+
+A [scenario](/docs/simulation/concepts/scenarios) holds the situation a conversation starts from, the flow it should follow, and a table of rows that each play out as their own conversation. You don't write those rows by hand. You point Future AGI at a source, say how many cases you want, and it generates them for you to edit.
+
+
+Scenarios are built against an [agent definition](/docs/simulation/concepts/agent-definitions), so create that first if you haven't. [Connect your agent](/docs/simulation/guides/connect-your-agent) walks through it. Personas need no setup: the 18 built-in [personas](/docs/simulation/concepts/personas) ship with every workspace, so there is always a set to attach.
+
+
+## Name the scenario and pick the agent
+
+Under **Simulate** in the sidebar, open **Scenarios** and click **Add Scenario**. The list holds every scenario in the workspace, with the agent type it targets, how many datapoints it carries, and whether generation has finished.
+
+
+*Every scenario in the workspace lives here*
+
+- **Choose source** and **Choose version**: The agent definition to build against, and the version to read. Pick these first
+- **Scenario Name**: Fills itself in from the two above, so `support-agent-chat` at `v1` becomes `support-agent-chat_v1`. Overwrite it if you'd rather name it yourself
+- **No. of scenarios**: This doesn't create 20 scenarios, it creates **one** scenario holding 20 rows. Each row is one conversation the simulator will run, and the Scenarios list reports the total as that scenario's datapoint count. The field accepts 10 to 20,000, so 10 rows is the smallest scenario you can generate
+
+
+*The name is derived from the agent and version you pick*
+
+Below these fields sits a row of four tabs, **Workflow builder**, **Import datasets**, **Upload script**, and **Call / Chat SOP**. Everything under the tabs belongs to the same form: pick a source there, then keep scrolling to the settings that follow.
+
+## Pick where the rows come from
+
+Each tab is a different source Future AGI can generate from. Pick the one that matches the material you already have.
+
+### Workflow builder
+
+The default, and the one to take when you have nothing to import. **Auto Generate Graph** is on, which means Future AGI drafts the conversation flow itself from your agent definition and its description, then writes the rows against that flow. For a first scenario this is usually all you need.
+
+
+*With Auto Generate Graph on, the flow is drafted for you*
+
+Turn **Auto Generate Graph** off and a **Manually Create Workflow** button appears, opening the visual graph builder so you can draw the flow yourself before generating. It stays disabled until you've chosen an agent definition, since the builder needs to know what it's building against. The builder is the same canvas you get on a finished scenario, and [Explore scenario graph](/docs/simulation/guides/explore-scenarios/scenario-graph) documents how to work in it.
+
+### Import datasets
+
+Builds the rows from data you already hold in a [dataset](/docs/dataset), so reach for it when your cases come from real material, like a spreadsheet of past tickets. Select the dataset and its rows become the cases.
+
+The dataset has to meet three conditions, and creation is rejected with the reason if it doesn't:
+
+- **At least 10 rows.** The error names the count it found, so a 6-row dataset fails before anything is generated
+- **No duplicate column names**
+- **A `persona` column, if present, must be typed as Persona.** A column literally named `persona` holding plain text is rejected; change its type in the dataset first
+
+
+*Only datasets in this workspace appear in the dropdown*
+
+### Upload script
+
+For when the conversation is already written down: a call script, a worked example dialogue, the wording your team is expected to follow turn by turn. Future AGI reads the document and builds the flow to match what it describes.
+
+
+*A script describes the conversation itself, turn by turn*
+
+### Call / Chat SOP
+
+For when what you have is the procedure rather than the dialogue: the policy your support team follows, its steps, conditions, and escalation rules. The generator turns those rules into cases that exercise them, which is the tab to pick when your written material says what must happen rather than what to say.
+
+
+*An SOP describes the rules; the generator writes conversations that test them*
+
+Both upload tabs accept **`.txt` and `.pdf` only**, and both read the file as text, so a PDF that is really a scan of a printed page gives the generator nothing to work with.
+
+## Generate from the agent definition, or your own instructions
+
+The settings from here down sit below the tabs on the same form, and they apply whichever source you picked.
+
+**Use only agent definition to create scenarios** is on by default, which keeps generation grounded in how your agent is configured. Turn it off and an **Extra Instruction** field appears, where you write additional instructions for the model to follow while generating. That's the way to steer the batch toward cases the agent definition alone wouldn't suggest, like a specific edge case you keep seeing in production.
+
+
+*Leave it on to generate from the agent definition, off to add your own instructions*
+
+## Attach the personas
+
+**Add by default** attaches every active [persona](/docs/simulation/concepts/personas) in the workspace to the scenario it generates. This is where personas enter a simulation: they ride along inside the scenario, so there's nothing to pick when you later start a run.
+
+
+*Personas attach here, which is why a run never asks you to choose one*
+
+Turn it off and an **Add persona** button appears, so you can attach a narrower set yourself. Like the graph builder, it needs an agent definition chosen first.
+
+## Add columns
+
+Every generated row already carries five columns: **persona**, **situation**, **outcome**, **conversation branch**, and **branch category**. **Columns** is for anything beyond those, up to ten of your own, named by you and used by the generator to vary the cases it writes. Add one when the cases differ along an axis the agent definition doesn't describe, such as a refund amount or a plan tier. You can also add columns later, from the scenario itself.
+
+## Create it
+
+Click **Create**. Generation runs in the background: the scenario appears in the list as **Running**, and you can leave the page while it works. It flips to **Completed** when the rows are ready.
+
+If it comes back **Failed**, look first at the material you imported rather than the form. A scanned PDF with no text layer and a dataset whose columns don't meet the conditions above are the common causes. If it completes but the rows read as weak, you don't have to start over: open the scenario and delete or replace the poor rows, or add better ones by hand.
+
+## Check what came out
+
+Open the scenario to read the flow it drafted and the rows it generated, and to change either. [Explore scenarios](/docs/simulation/guides/explore-scenarios) covers the graph, adding rows, and adding columns.
+
+## Dive deeper
+
+
+
+ Read and edit the flow and rows you just generated
+
+
+ Build a custom customer for your scenarios to carry
+
+
+ Put the scenarios in front of your agent
+
+
diff --git a/src/pages/docs/simulation/guides/create-simulation.mdx b/src/pages/docs/simulation/guides/create-simulation.mdx
new file mode 100644
index 00000000..f93947b2
--- /dev/null
+++ b/src/pages/docs/simulation/guides/create-simulation.mdx
@@ -0,0 +1,173 @@
+---
+title: "Create a simulation"
+description: "Bundle an agent version, scenarios, and evals into a run, then start it"
+---
+
+A simulation pairs one agent version with the [scenarios](/docs/simulation/concepts/scenarios) it has to face and the evals that score what comes back. The dashboard calls it a simulation; the docs and the SDK call the same object a [run test](/docs/simulation/concepts/runs-and-results).
+
+Building one is a four-step wizard, and the four steps are identical whether your agent talks or types. Only what happens after you create it splits by channel: a voice run places its own calls, while a chat run waits for you to drive it from your own code.
+
+
+Two things have to exist first: an [agent definition](/docs/simulation/concepts/agent-definitions) with at least one version, from [Connect your agent](/docs/simulation/guides/connect-your-agent), and a scenario built against it, from [Create scenarios](/docs/simulation/guides/create-scenarios). [Personas](/docs/simulation/concepts/personas) need nothing here: they ride along inside the scenario, which is why the wizard never asks you to pick one. A chat run needs one more thing, an [API key pair](/docs/admin-settings/api-keys), but only at the end, when you run it from your own code.
+
+
+## Name the run and pick the agent
+
+Under **Simulate** in the sidebar, open **Run Simulation**. The list holds every run in the workspace, with the agent it targets, the scenarios and evals attached to it, and when it last ran. Click **Create a Simulation** to open the wizard.
+
+
+*A run's row carries its scenarios and evals, so the list doubles as a record of what was tested*
+
+The first step, **Add simulation details**, asks for four things:
+
+- **Simulation name**: required, and you type it yourself. Nothing generates it for you. Pick something you'll still recognise in a list six weeks from now, because a chat run's code references this name exactly
+- **Choose Agent definition**: required, and the choice that shapes the rest of the wizard. It sets the channel, which decides both the scenarios you can pick and how the run starts
+- **Choose version**: required, and disabled until a definition is chosen. It defaults to the newest version, so choose an older one deliberately, when you're re-running a configuration you've since edited past
+- **Description**: optional. Worth a line anyway, since it's what tells you months later why this run existed
+
+
+*Choose version stays disabled until a definition is picked, since versions belong to one*
+
+The definition and version are worth a second look before moving on: neither can be changed once the run is created.
+
+## Choose the scenarios
+
+The second step lists the scenarios in the workspace that match your agent's channel: pick a chat agent and only chat scenarios appear. Tick as many as you want, and at least one is required to move on. Like the agent version, the set you tick here is fixed once the run exists.
+
+
+*A scenario with no rows is greyed out and can't be ticked*
+
+The number on the right is the scenario's row count, and it's what sets the size of the run. A run against a 20-row scenario plays 20 conversations, one per row. Tick two scenarios and it plays both sets.
+
+A scenario showing **0** has no rows to run against. Open it from **Scenarios** and [add rows](/docs/simulation/guides/explore-scenarios/add-rows) first.
+
+## Add the evals
+
+The third step is where you decide what counts as good. Nothing is scored by default, and the step won't let you past until at least one eval is on the run.
+
+
+*The tool call switch sits outside the eval list and is off until you turn it on*
+
+Click **Add Evaluations** to open the library.
+
+**Enable tool call evaluation**, above it, is a separate switch and optional. Turn it on when your agent calls tools and you want the calls themselves checked, not only what the agent said. [Evaluate tool calls](/docs/simulation/guides/evaluate-tool-calls) covers what that scores.
+
+### Pick one from the library
+
+The drawer lists the eval library. Search by name, or narrow a long list with the category chips, then click **Add** on the one you want.
+
+
+*Add opens the eval's configuration rather than attaching it straight away*
+
+### Point it at the right field
+
+An eval already knows how to judge. What it doesn't know is which part of the conversation to read, and **Variable Mapping** at the bottom of its configuration is where you tell it: each of the eval's inputs gets a dropdown, and you pick the column that feeds it.
+
+Most evals want `call.transcript`, the whole conversation, and that's the sensible default. Reach for something narrower when the eval only makes sense against one side, like scoring your agent's tone from `call.assistant_chat_transcript`, or when it needs audio rather than text.
+
+The full set of columns is listed above the mapping:
+
+- **The call**: `call.transcript`, plus `call.agent_prompt`, `call.duration_seconds`, `call.status`, and `call.overall_score`
+- **Chat only**: `call.user_chat_transcript` and `call.assistant_chat_transcript`, which hold one side of the conversation each
+- **Voice only**: `call.summary`, and the recordings `call.voice_recording`, `call.assistant_recording`, `call.customer_recording`, and `call.stereo_recording`, for evals that listen rather than read
+- **Context**: `scenario`, `persona`, `simulation`, and `agent` fields describing what the call was set up to do
+
+
+*Values stay `` until a run has produced them*
+
+Built-in evals arrive pre-configured, so their instructions and output type are shown for reference and can't be changed. The mapping is the part you set. Click **Add Evaluation** and you land back on the step.
+
+
+*Add More stacks another eval onto the same run*
+
+The name is the giveaway. `tone_simulation_07_aug_2026_10_45` is not the library's `tone` eval, it's a copy stamped with the date and bound to this run, which is what the step's banner means by "Selected evaluations will be created and linked to this simulation run". Retune its mapping and nothing changes for anyone else using `tone`.
+
+## Review and create
+
+The last step lays the bundle out in one scroll: name and description, agent definition and version, every scenario with its row count, and every eval with its mapping.
+
+
+*The Summary step is the whole bundle in one place*
+
+This is the last chance to change the two choices that are one-way. A created run has no edit, only **View** and **Delete** in its row menu, so a different agent version or a different set of scenarios means building another run. Evals are the exception: the **Evals** chip on the run's own page adds and removes them afterwards, which is what [Edit evals in a simulation](/docs/simulation/guides/edit-evals) covers.
+
+Click **Run Simulation** to create it. Despite the label, whether anything actually runs now depends on the channel.
+
+## Start the run
+
+Read the half that matches your agent; the other doesn't apply.
+
+### Voice runs start on their own
+
+A voice run begins the moment it's created. Future AGI places the calls itself, so there's nothing to install and nothing to run on your machine. You land on the run's **Simulated runs** tab, and the run appears there as an execution that reports its own progress, moving through **pending**, **running**, and **evaluating** before it settles on **completed**, or on **failed** if it stopped early. Calls fill in underneath as they finish, so a run mid-flight shows some of its rows rather than none.
+
+**Run New Simulation** plays the same bundle again as a fresh execution, which is how you compare two attempts at identical settings.
+
+
+*A voice run executes on the platform; this tab fills in as calls complete*
+
+### Chat runs wait for your code
+
+A chat agent lives in your code, where Future AGI can't reach it, so creating the run starts nothing. The **Simulated runs** tab hands you the boilerplate to drive it yourself, already carrying this run's name.
+
+
+*Copy from this panel rather than from the page below: its `run_test_name` is already your run's*
+
+Install the SDK:
+
+```bash
+pip install agent-simulate
+```
+
+`TestRunner` reads your [API key pair](/docs/admin-settings/api-keys) from `FI_API_KEY` and `FI_SECRET_KEY`, so set both in your environment. Then point `run_test` at the function that answers a message in your app:
+
+```python
+import asyncio
+from fi.simulate import TestRunner, AgentInput
+
+async def customer_support_agent(input: AgentInput) -> str:
+ user_message = input.new_message["content"] if input.new_message else ""
+ # Call your own agent here and return what it says
+ return await my_agent.respond(user_message)
+
+async def main():
+ runner = TestRunner()
+ report = await runner.run_test(
+ run_test_name="Simulating support-agent-chat", # your run's name, exactly
+ agent_callback=customer_support_agent,
+ )
+ print(f"Processed {len(report.results)} test cases")
+
+asyncio.run(main())
+```
+
+Two things have to be right. `run_test_name` must match the run's name character for character, which is why copying from the dashboard's panel is safer than retyping. And the callback runs once per turn, receiving `new_message` for the turn to answer, `messages` for the conversation so far, and `thread_id` identifying the conversation; return the reply as a string.
+
+Run the script. Future AGI plays each scenario row as a customer, your callback answers each turn, and the transcripts come back to the run. [Run a chat simulation](/docs/simulation/guides/run-chat-simulation) goes through the callback in full, including returning an `AgentResponse` instead of a string when you want your agent's tool calls reported alongside the reply.
+
+A chat run that's never driven simply stays empty; it isn't queued anywhere and it won't time out, so an untouched **Simulated runs** tab means the script hasn't run, not that the run failed. The boilerplate stops showing once the run has its first execution, and **Run New Simulation** brings it back when you need it again.
+
+## Read the results
+
+Once calls exist, the two tabs beside **Simulated runs** are where they land. **Call Details**, titled **Chat Details** on a chat run, holds one row per call, so a 20-row scenario leaves 20 rows, each with its status and its transcript. **Analytics** aggregates the same calls into eval scores across the run, which is what you compare when you run the bundle a second time. [Explore results](/docs/simulation/guides/explore-results) covers both.
+
+## Dive deeper
+
+
+
+ Wire your agent into the SDK and play the scenarios
+
+
+ Take a voice agent through the same bundle
+
+
+ What a run leaves behind to read and compare
+
+
diff --git a/src/pages/docs/simulation/guides/edit-evals.mdx b/src/pages/docs/simulation/guides/edit-evals.mdx
new file mode 100644
index 00000000..e75370b1
--- /dev/null
+++ b/src/pages/docs/simulation/guides/edit-evals.mdx
@@ -0,0 +1,54 @@
+---
+title: "Edit evals in a simulation"
+description: "Add, update, remove, or remap the evals scoring a run that already exists"
+---
+
+A run's evals aren't fixed the way its agent version and scenarios are. Add one you forgot, retune where one reads from, or drop one that isn't earning its place, all from the run's own page, long after the run was created.
+
+
+This page assumes a run test that already exists, from [Create a simulation](/docs/simulation/guides/create-simulation). Picking which eval to add and what it measures is [Evaluation](/docs/evaluation)'s territory; this page only covers attaching, editing, and rerunning evals already on a run.
+
+
+## Open the run's evals
+
+Under **Simulate** in the sidebar, open **Run Simulation** and click into the run. The **Evals** chip on its page opens the **All Evaluations** panel, which lists every eval currently attached, each row carrying its own edit and delete controls.
+
+**Enable tool call evaluation** doesn't live in this list, it's a checkbox at the bottom of this same panel. Turn it on or off from there and it saves straight onto the run. [Evaluate tool calls](/docs/simulation/guides/evaluate-tool-calls) covers what it scores.
+
+## Add an eval
+
+Click **Add Evaluation** if the run has no eval attached yet, or **Add** in the list header if it already does, to open the same library used when the run was created. Pick an eval, then set its **Variable Mapping** to the column it should read from, the same mapping step covered in [Create a simulation](/docs/simulation/guides/create-simulation). Save and it's appended to the run's list, named after the eval and stamped with the date, exactly like the ones added when the run was built.
+
+## Update an eval, or remap its variables
+
+Click the edit icon on any eval in the list to reopen the configuration it was added with. What's editable depends on the eval: a built-in eval only exposes its mapping, since its instructions and output type are fixed, while an eval with its own settings, like a filter or a knowledge base, exposes those too, and you can even swap it for a different eval outright by changing its template.
+
+Remapping is the one edit every eval takes, built-in or not: open **Variable Mapping** and point any input at a different column than the one it currently reads, then save.
+
+## Remove an eval
+
+Click the delete icon on the eval's row and confirm. It's gone from the run. A run always needs at least one eval, so removing it is blocked while it's the only one left; add a replacement first if you're swapping it out rather than dropping it.
+
+## Get updated scores without rerunning calls
+
+None of the edits above touch a call that's already run: it keeps whatever eval scores it got at the time. To see how the current eval configuration would have scored it instead, rerun evals rather than the whole run.
+
+From the run's page, click **Re-run simulation** and choose **Run Evals**. It's the only option a chat run offers, since a chat agent's calls live in your own code and can't be replayed by the platform, and it's also the option worth reaching for on a voice run when only the scoring changed: every call's recording, transcript, and cost stay exactly as they were, only its eval outputs are cleared and recalculated against whatever evals are on the run now. If the rerun fails before it starts, nothing about the calls changes either. [Run a voice simulation](/docs/simulation/guides/run-voice-simulation) covers the confirmation step and the alternative, **Run test + Evals**, which replaces the calls themselves.
+
+
+**Re-run simulation** doesn't appear at all on a run built from a Prompt Workbench prompt, and it's disabled until the run has at least one completed call.
+
+
+## Dive deeper
+
+
+
+ Trigger Run Evals or Run test + Evals, and confirm before either starts
+
+
+ Score the tool calls a run made, kept separate from these evals
+
+
+ What a run keeps and what a rerun replaces
+
+
diff --git a/src/pages/docs/simulation/guides/evaluate-tool-calls.mdx b/src/pages/docs/simulation/guides/evaluate-tool-calls.mdx
new file mode 100644
index 00000000..73230d18
--- /dev/null
+++ b/src/pages/docs/simulation/guides/evaluate-tool-calls.mdx
@@ -0,0 +1,46 @@
+---
+title: "Evaluate tool calls"
+description: "Turn on tool call scoring for a run test, and see where the results land"
+---
+
+Tool call evaluation scores the tool calls your agent made during a conversation, separately from the evals that score what it said. Turn it on for a [run test](/docs/simulation/concepts/runs-and-results) and each call gets its own tool-call results, kept apart from the rest of that call's eval scores rather than folded into them.
+
+## Turn it on
+
+The switch lives on the **Select evaluations** step of the [simulation wizard](/docs/simulation/guides/create-simulation), labelled **Enable tool call evaluation**. It sits above the eval library, off by default, and turning it on doesn't count as one of the evals that step requires you to add. It's set once, when you create the run.
+
+Flip it only for a run test whose agent actually calls tools during the scenarios you've attached. A scenario that never reaches a tool leaves nothing for it to evaluate.
+
+
+Tool call evaluation only works for agents on **Vapi**. Turning it on for a chat agent, or for a voice agent on Retell or Bland.ai, leaves nothing to evaluate, even though both are supported [voice providers](/docs/simulation/reference/voice-providers) for the rest of Simulation.
+
+
+## Report tool calls from a chat agent
+
+Vapi is the only place tool call evaluation actually scores anything, so a chat agent's tool calls aren't evaluated even when you report them. The shape is still the one your callback needs to use if you want tool calls to show up at all: return an `AgentResponse` instead of a plain string, and set two of its fields: `tool_calls`, the tools your agent decided to call, and `tool_responses`, the results that came back from them, each entry a dict with `role`, `tool_call_id`, and `content`. If your agent already holds the raw tool output in a different shape, pass it through `metadata={"tool_outputs": [{"call_id": ..., "output": ...}]}` instead and the SDK converts it for you.
+
+[Run a chat simulation](/docs/simulation/guides/run-chat-simulation) covers the full callback contract, including the plain-string return you use when tool calls aren't part of what you're testing.
+
+## Voice calls need nothing returned from you
+
+Unlike a chat agent, a voice agent on Vapi doesn't need to return anything for its tool calls to be evaluated.
+
+## Where the results show up
+
+Open a call from the run's results and its tool-call results sit alongside that call's other eval scores, as their own entry rather than mixed into them. [Calls & transcripts](/docs/simulation/guides/explore-results/calls-and-transcripts) walks through that view in full.
+
+The transcript itself won't help here: it's built from what the persona and the agent said to each other, so a tool call never shows up as a turn in it. Check the tool-call results for that call instead of scanning the transcript for what got called.
+
+## Dive deeper
+
+
+
+ Read a call's transcript, metrics, and eval results together
+
+
+ The full AgentResponse contract, including tool_calls and tool_responses
+
+
+ What each voice provider supports in Simulation
+
+
diff --git a/src/pages/docs/simulation/guides/explore-results/analytics.mdx b/src/pages/docs/simulation/guides/explore-results/analytics.mdx
new file mode 100644
index 00000000..dfc7b65d
--- /dev/null
+++ b/src/pages/docs/simulation/guides/explore-results/analytics.mdx
@@ -0,0 +1,48 @@
+---
+title: "Analytics & metrics"
+description: "Read the performance summary, KPIs, and eval summary on a run's Analytics tab"
+---
+
+A run test's **Analytics** tab rolls every call in the run into three reads: a performance summary, a set of KPI cards, and an eval summary. None of them replace opening an individual call, covered in [Calls & transcripts](/docs/simulation/guides/explore-results/calls-and-transcripts); they're for judging the run as a whole before you drop into one conversation.
+
+## Read the performance summary
+
+One read is the performance summary, **Test Run Performance Metrics**, three cards:
+
+- **Pass Rate**: the share of calls in the run that came back passing. A number here below what you'd expect from a stable agent is the first sign something regressed
+- **Total Test Runs**: how many times this run test has been executed. One card, not per-call, so it tells you how many attempts you have to compare, not how any single attempt went
+- **Latest Fail Rate**: the fail rate of the most recent execution specifically. Read it next to Pass Rate to tell a run that's always been shaky from one that just got worse
+
+Alongside the cards, **Top Performing Scenarios** lists each scenario with a score chip: red under 5, orange under 7, green at 7 and above. A scenario sitting in red is where the agent is struggling hardest, and it's the one worth opening first.
+
+## Read the KPIs
+
+The KPI cards differ by channel, because a voice call and a chat conversation are measured differently.
+
+A voice run shows CSAT (the header label for the call's overall score), Agent Latency in milliseconds, the agent's words-per-minute pace, how quickly the agent stops talking when the caller cuts in, average turn count, and a talk ratio comparing how much of the call the agent spent talking against the caller. A low CSAT alongside high latency or a lopsided talk ratio usually points at the same root cause: the agent is talking too much, or too slowly, to keep the caller satisfied.
+
+A chat run shows CSAT again, average latency in milliseconds, average turn count, and three token counts: total, input, and output. Token counts matter here beyond cost. A run whose output tokens climb without a matching lift in CSAT is spending more per reply without the conversation actually getting better.
+
+Every field behind these cards, including the ones not surfaced as KPI cards, is listed in [Call metrics](/docs/simulation/reference/call-metrics).
+
+## Read the eval summary
+
+The eval summary is one card per eval attached to the run, graphing how that eval scored across every call. This is where you catch an eval that's consistently weak across the whole run rather than failing on one unlucky call, and it's the signal that tells you which eval to chase into individual transcripts. Eval types and how each one is configured belong to [Evaluation](/docs/evaluation); this tab only shows the scores it produced.
+
+
+Rerunning a call, from [Calls & transcripts](/docs/simulation/guides/explore-results/calls-and-transcripts), updates the eval summary and KPIs the next time you load this tab.
+
+
+## Dive deeper
+
+
+
+ Open one call to read its transcript and per-call scores
+
+
+ Every metric a call carries, field by field
+
+
+ Turn a weak eval summary into concrete fixes
+
+
diff --git a/src/pages/docs/simulation/guides/explore-results/calls-and-transcripts.mdx b/src/pages/docs/simulation/guides/explore-results/calls-and-transcripts.mdx
new file mode 100644
index 00000000..4c77c562
--- /dev/null
+++ b/src/pages/docs/simulation/guides/explore-results/calls-and-transcripts.mdx
@@ -0,0 +1,50 @@
+---
+title: "Calls & transcripts"
+description: "Open one call and read its transcript, evals, cost, and recording"
+---
+
+A [run test](/docs/simulation/concepts/runs-and-results) fans out into calls, one per scenario row, and each call is the full record of one conversation. This guide opens a single call and reads through everything it carries: the transcript, the recording, the evals it scored, and what it cost.
+
+## Open a call
+
+Every call is a row in the run's **Call Details** tab (**Chat Details** for a chat run). Click a row and its detail drawer opens over the grid.
+
+The drawer opens on **Call Log Details**, with chips for the [persona](/docs/simulation/concepts/personas) or customer name, the [scenario](/docs/simulation/concepts/scenarios) the call ran, when it started, how long it ran, and a status badge. A **View Docs** link in the header points back to the simulation docs.
+
+## Follow the transcript
+
+The transcript runs turn by turn, and each turn carries a speaker role. Three of them show up in the transcript you read: `USER` is the simulated persona's turn, `ASSISTANT` is your agent's, and `SYSTEM` is a turn that came from a system-level instruction rather than either side of the conversation.
+
+If your agent calls tools mid-conversation, those turns exist too, under two further roles kept out of the transcript view: one holds the name of the tool that was called, the other the result it returned. They aren't something you'd otherwise see here; [Evaluate tool calls](/docs/simulation/guides/evaluate-tool-calls) covers scoring them directly.
+
+## Play back the recording
+
+A voice call's drawer docks a recording player next to the transcript, so you can listen to the call while you follow what was transcribed from it. Chat calls carry no audio: the transcript is the whole record.
+
+## Check the evals scored on this call
+
+Every [eval](/docs/evaluation) attached to the run test scores each call independently, and the drawer lists all of them with the score this specific call got. This is the per-call view; the run-wide totals live on [Analytics & metrics](/docs/simulation/guides/explore-results/analytics) instead, and every field a call can carry, evals included, is defined exhaustively in the [Call metrics reference](/docs/simulation/reference/call-metrics).
+
+The same drawer can rerun this one call: a voice call offers **Run Evals** or **Run test + Evals**, a chat call offers **Run Evals** only.
+
+## See what it cost
+
+The **Cost** section totals what this call cost and, where there's something to break down, splits the total into categories such as speech-to-text, the language model, text-to-speech, and recording storage. A call with nothing to show here reads as "No additional details available" rather than a zero.
+
+
+The drawer also carries **Compare with baseline**, which opens the original-versus-replay comparison. [Replay](/docs/simulation/concepts/replay) covers how that comparison works.
+
+
+## Dive deeper
+
+
+
+ Aggregate scores across every call in a run
+
+
+ Score the tool calls a transcript keeps out of view
+
+
+ Turn a run's failing calls into an improved agent
+
+
diff --git a/src/pages/docs/simulation/guides/explore-results/index.mdx b/src/pages/docs/simulation/guides/explore-results/index.mdx
new file mode 100644
index 00000000..68e7652f
--- /dev/null
+++ b/src/pages/docs/simulation/guides/explore-results/index.mdx
@@ -0,0 +1,64 @@
+---
+title: "Explore results"
+description: "Find your way around a run's results page: the header, its tabs, and where each one leads"
+---
+
+Once a run has calls in it, whether it's still going or long finished, everything it produced collects on one page. This page maps that layout: the header, the tabs, and what each tab is for, so the two guides after it can go straight to the detail without re-explaining where things live. If you don't have a run to look at yet, [Create a simulation](/docs/simulation/guides/create-simulation) builds and starts one first.
+
+What you're looking at is one [execution](/docs/simulation/concepts/runs-and-results) of a run, not a log of every attempt. A run can be started more than once, and each attempt gets its own copy of this page with its own calls and its own scores.
+
+## Open a run's results
+
+Under **Simulate** in the sidebar, **Run Simulation** lists every run in the workspace. Click a row to open it, or land here directly right after creating one, as Create a simulation walks through.
+
+## The page at a glance
+
+ H["Header name, status, export, rerun, stop"]
+ R --> T1["Call Details / Chat Details one row per call"]
+ R --> T2["Analytics aggregated eval scores"]
+ R --> T3["Optimization Runs optimization attempts"]
+ T1 -.-> G1(["Calls and transcripts guide"])
+ T2 -.-> G2(["Analytics and metrics guide"])
+`} />
+
+## The header
+
+The header names the run and carries the actions that apply to the whole execution rather than to one call:
+
+- **Export Data** downloads every call in this execution as a CSV
+- **Re-run simulation** starts the calls again, or just their evals, depending on what you pick. It's hidden on runs [simulated from a prompt](/docs/simulation/guides/prompt-simulation) rather than an agent definition, and disabled until there's at least one call to rerun. The same control appears again at the call level, one call at a time
+- **Stop Running** appears only while the execution is active, and asks you to confirm before it cancels the rest of the calls
+
+The header also shows the execution's status as it moves from pending through running to completed (or failed, or cancelled), which [Runs & results](/docs/simulation/concepts/runs-and-results) covers in full.
+
+## The three tabs
+
+Below the header sit three tabs. Two of them route onward to their own guide; the third, optimization, has a lighter guide of its own too.
+
+### Call Details / Chat Details
+
+This is the tab you land on, and its title changes with the agent: **Call Details** for a voice run, **Chat Details** for a chat run. A row of summary cards sits at the top, then the grid underneath lists every call in this execution, one row each, with its status and its score if the run has evals attached.
+
+Click a row to open that call. That drawer, the transcript inside it, and the evaluation results per call are what [Calls and transcripts](/docs/simulation/guides/explore-results/calls-and-transcripts) walks through.
+
+### Analytics
+
+Analytics rolls the same calls up into one view: eval scores aggregated across the whole execution, rather than one call at a time. It's what you'd check to see how the run did overall. [Analytics and metrics](/docs/simulation/guides/explore-results/analytics) covers what's on it.
+
+### Optimization Runs
+
+If you've sent this execution's results through [Fix My Agent](/docs/simulation/guides/fix-my-agent), the attempts show up here with their type, trial count, and status. [Optimization runs](/docs/simulation/guides/optimization-runs) covers reading one.
+
+## Dive deeper
+
+
+
+ Open a call, read its transcript, and act on one result
+
+
+ Read how the whole execution performed, not just one call
+
+
diff --git a/src/pages/docs/simulation/guides/explore-scenarios/add-columns.mdx b/src/pages/docs/simulation/guides/explore-scenarios/add-columns.mdx
new file mode 100644
index 00000000..b527ee9f
--- /dev/null
+++ b/src/pages/docs/simulation/guides/explore-scenarios/add-columns.mdx
@@ -0,0 +1,82 @@
+---
+title: "Add columns"
+description: "Give a scenario's rows extra variables the simulator prompt can use"
+---
+
+A column is one variable each row of a [scenario](/docs/simulation/concepts/scenarios) carries, and the simulator prompt can read it by name. Add a `refund_amount` column and each conversation runs with its own row's amount.
+
+Every generated row already carries five: **persona**, **situation**, **outcome**, **conversation_branch**, and **branch_category**. A column you add is anything beyond those.
+
+
+This page adds columns to a scenario that already exists. If you haven't generated one yet, the create form carries the same Columns section, covered in [Create scenarios](/docs/simulation/guides/create-scenarios).
+
+
+## Open the column form
+
+Under **Simulate** in the sidebar, open **Scenarios** and click into your scenario. Above the **Generated scenarios** table, click **Add Column** to open the drawer.
+
+
+*The table already shows the five built-in columns the rows were generated with*
+
+## Define the column
+
+Choose who fills in the values. The form is the same either way, and so is the result: a new column on every row.
+
+- **Add Manually**: you type the values yourself. Right when the values matter exactly, a specific plan tier or a refund amount you're testing a threshold against, and when there are few enough rows to be worth typing
+- **Generate using AI**: the values are written from your description. Right when you want plausible variety across many rows rather than particular numbers, which is the common case on a 20-row scenario and the only practical one on a few hundred
+
+Each column takes three things, all required:
+
+- **Column name**: what the prompt will reference, so keep it short and lowercase, like `refund_amount`
+- **Data type**: seven to pick from. **Text** for anything wordy, **Integer** for whole numbers, **Float** for amounts with decimals, **Boolean** for a yes/no flag, **Date & Time** for a date. **JSON** and **Array** hold structured values, which read awkwardly once dropped into a sentence, so keep them out of the prompt and use them for data the flow reads instead
+- **Description**: what the column holds. On the AI path this is the instruction the values are generated from, so be specific: "the refund amount in dollars, between 20 and 500" beats "amount"
+
+
+*Both paths ask for the same three fields; only who fills the rows differs*
+
+Two rules the form enforces:
+
+- **Ten columns per pass.** Inside the drawer, **+ Add Column** defines another column in the same save, up to ten at once. Adding more than ten means opening the drawer again, there's no cap on the scenario itself
+- **Names must be new.** A name already on the scenario is rejected, so you can't reuse `persona`, `outcome`, or any column you added earlier, and two columns in the same pass can't share a name
+
+## What lands in the table
+
+Saving closes the drawer, shows a "Columns added successfully" message, and refreshes the table with the new column at the end. On both paths the column arrives **empty**, and what happens next differs:
+
+- **Manual**: it stays empty until you fill it. Type into each row's cell in the table itself, the way you would in a [dataset](/docs/dataset)
+- **AI**: generation runs in the background, so the cells fill in after a moment rather than the instant the drawer closes. Refresh the table if it still looks empty
+
+Nothing here is permanent. Cells stay editable after generation, so you can correct a value the model got wrong, and a column you mis-named can be deleted from its header menu in the table.
+
+## Use the column in the prompt
+
+Open **Prompt** on the scenario and reference the column by name in double braces:
+
+```text
+The customer is asking for a refund of {{refund_amount}}.
+```
+
+The braces are matched exactly, so `{{refund_amount}}` reads the `refund_amount` column and nothing else.
+
+## Fix a red variable
+
+The prompt colours its variables as a check:
+
+- **Green**: the name matches a column on this scenario, so it will be filled
+- **Red**: nothing will fill it
+
+A red variable is almost always a misspelled name or a column that was never added. Compare it against the column headers in the table and fix whichever is wrong: correct the spelling in the prompt, or add the missing column. The variable turns green once the two match.
+
+## Dive deeper
+
+
+
+ Add more test cases for your columns to describe
+
+
+ Read and edit the flow these rows run through
+
+
+ Put the scenario in front of your agent
+
+
diff --git a/src/pages/docs/simulation/guides/explore-scenarios/add-rows.mdx b/src/pages/docs/simulation/guides/explore-scenarios/add-rows.mdx
new file mode 100644
index 00000000..3101bf92
--- /dev/null
+++ b/src/pages/docs/simulation/guides/explore-scenarios/add-rows.mdx
@@ -0,0 +1,51 @@
+---
+title: "Add rows"
+description: "Put more test cases in a scenario, generated from a description or typed by hand"
+---
+
+Rows are the individual test cases a [scenario](/docs/simulation/concepts/scenarios) holds: 20 of them on `support-agent-chat_v1`, each playing out as its own conversation through the same flow. When the generated set doesn't cover a case you care about, you add rows to it.
+
+This picks up from a scenario you already have open, so start at [Explore scenarios](/docs/simulation/guides/explore-scenarios) if you need one in front of you first.
+
+## Open the Add Rows panel
+
+**Add Row** sits above the **Generated scenarios** table on the right, and opens the **Add Rows** panel.
+
+
+*Describe the cases you want, or type them in yourself*
+
+## Generate rows with AI
+
+Pick this when you can describe the cases but don't want to write them. **No.of rows** takes a count between 10 and 20,000, and **Description** is where you say what the rows should cover, something like `customers disputing a charge over $200 who have already contacted support twice`. Click **Add** and the new rows arrive with their columns filled in, ready to edit like the generated ones.
+
+
+*Describe the cases and the generator writes the rows*
+
+## Add empty rows
+
+Pick this when you already have the cases in mind and just need somewhere to put them. **No. of rows to create** takes a number from 1 to 10, and **Next** adds that many blank rows for you to type into.
+
+
+*Blank rows are the path for cases you already know*
+
+## Pull rows from a dataset, when the scenario has one
+
+Scenarios built from a dataset carry one more route, sitting above the other two in the panel: **Add from existing model dataset or experiment**. Pick it, choose what you want under **Choose Datasets or experiments**, and click **Add** to copy those rows into the scenario table. It's the fastest path when the cases you want to test are already recorded in a [dataset](/docs/dataset) you hold. A scenario built any other way doesn't show this option at all.
+
+## Remove rows
+
+Select the rows you don't want and the buttons above the table swap for a selection bar carrying the count and a **Delete** action.
+
+## Dive deeper
+
+
+
+ Give the simulator prompt something new to vary on
+
+
+ Build a custom customer for these rows to carry
+
+
+ Put the rows you just added in front of your agent
+
+
diff --git a/src/pages/docs/simulation/guides/explore-scenarios/index.mdx b/src/pages/docs/simulation/guides/explore-scenarios/index.mdx
new file mode 100644
index 00000000..8c1522d9
--- /dev/null
+++ b/src/pages/docs/simulation/guides/explore-scenarios/index.mdx
@@ -0,0 +1,53 @@
+---
+title: "Explore scenarios"
+description: "Find your way around a generated scenario: its flow, its simulator prompt, and its rows"
+---
+
+A generated [scenario](/docs/simulation/concepts/scenarios) is a first draft, not a finished test suite. Its detail view is where you read the flow Future AGI drafted, check the prompt the simulator will follow, and change the cases it will run. These guides walk that view on `support-agent-chat_v1`, a chat scenario with 20 datapoints built on a customer-support agent.
+
+They all start from a scenario you have already generated, so if you don't have one yet, [Create scenarios](/docs/simulation/guides/create-scenarios) makes the first one.
+
+## Open the scenario
+
+Go to **Scenarios** under **Simulate** in the sidebar and click the `support-agent-chat_v1` row. The whole row is the target, so there's no separate open action to find. If there's nothing in the list yet, [Create scenarios](/docs/simulation/guides/create-scenarios) generates the first one.
+
+
+*Clicking anywhere on the row opens that scenario*
+
+## Read the three regions of the detail view
+
+The header carries the scenario name under an **All Scenarios** breadcrumb, plus four facts about it: **Agent Type**, **Scenario Type**, **No of Datapoints**, and **Created**. `support-agent-chat_v1` reads Chat, Graph, 20, and however long ago it was generated. **No of Datapoints** counts the rows in the table below, so one datapoint is one row is one test case, and those three names all point at the same thing.
+
+Below the header the view splits into three regions:
+
+- the conversation graph on the left, the flow every conversation follows
+- the **Prompt** panel on the right, the simulator prompt that plays the customer
+- **Generated scenarios** at the bottom, the table of rows
+
+Read the **Prompt** panel first even though it sits in the middle of that list, because it reaches into both of the others.
+
+
+*The flow, the prompt, and the rows, all on one page*
+
+## Why the Prompt panel ties them together
+
+The prompt is one instruction the simulator runs for every row, and it pulls each row's values in through `{{variable}}` placeholders, so `{{situation}}` in the prompt becomes that row's situation. Two things follow from that:
+
+- **Placeholders are colour-checked.** Green means a column of that name exists in the table, red means it doesn't, which is the fastest way to spot a prompt reaching for a column that was never created
+- **`{{` opens a picker.** Click **Edit** to rewrite the prompt, and type `{{` in the editor to choose from the table's columns rather than spelling a name out
+
+So the graph decides how a conversation moves, the rows decide what varies between conversations, and the columns are what the prompt is allowed to read. That coupling is what the rest of these guides act on.
+
+## Dive deeper
+
+
+
+ Read and edit the flow every conversation follows
+
+
+ Put more test cases in front of your agent
+
+
+ Give the simulator prompt something new to vary on
+
+
diff --git a/src/pages/docs/simulation/guides/explore-scenarios/scenario-graph.mdx b/src/pages/docs/simulation/guides/explore-scenarios/scenario-graph.mdx
new file mode 100644
index 00000000..358dbd40
--- /dev/null
+++ b/src/pages/docs/simulation/guides/explore-scenarios/scenario-graph.mdx
@@ -0,0 +1,70 @@
+---
+title: "Explore scenario graph"
+description: "Read and edit the conversation flow your scenario runs, node by node"
+---
+
+The conversation graph is the flow every row of a [scenario](/docs/simulation/concepts/scenarios) plays out: each node is a step your agent takes, and each edge is the condition that moves the conversation on.
+
+Open a scenario from **Scenarios** under **Simulate** and the graph fills the left half of its detail view, laid out for reading: pan and zoom it with the controls at its bottom left, and read any step's wording straight off the node's card.
+
+Editing the graph changes what the simulation actually tests, so reach for the builder when the drafted flow doesn't match the agent you're testing: a branch it handles that the graph never offers, a conversation that ends before it should, or a step whose wording sends the customer down the wrong path.
+
+This guide works on `support-agent-chat_v1`, the chat scenario from the [Explore scenarios overview](/docs/simulation/guides/explore-scenarios).
+
+## Open the graph editor
+
+On that detail view, click **Edit** at the top right of the graph and the full-screen **Flow Builder** opens: the canvas in the middle, a palette of node types down the left, and **Save flow** above the palette.
+
+Nothing you do in the builder is stored until you click **Save flow**, and closing with unsaved edits asks whether to save or discard them first.
+
+
+*Edit opens the Flow Builder over the whole page*
+
+## Inspect a node
+
+Each node card on the canvas already shows the essentials: its type, its **Prompt**, a **Start** label on the node the conversation begins at, and a **Global** chip on a node the agent can reach from any point in the flow. Click a node and it becomes the active one, with its detail panel opening on the right under the node's name.
+
+
+*Clicking a node highlights it on the canvas and opens its panel*
+
+The panel is where a step's wording lives. A Conversation node carries **Node Type**, the **Prompt** that step runs on, and an **Enable Global Node** toggle for making it reachable from anywhere. Edit the prompt in place and the node card on the canvas updates as you type; **Save flow** is still what writes it back.
+
+
+Leave **Node Type** alone unless you really mean to change what the step is. Switching it resets the node to the defaults for the new type, and the prompt you wrote goes with it.
+
+
+
+*A node's prompt is edited in the panel, not on the card*
+
+Edges work the same way. Click the line between two nodes and a **Condition** panel opens, holding the condition that sends the conversation down that branch.
+
+## Add a node by dragging it in
+
+New nodes come from the palette on the left, and they're dragged rather than clicked: pick a type up and drop it where you want it on the canvas. The palette follows the agent type, so a chat scenario like `support-agent-chat_v1` offers **Conversation**, **End chat**, and **Transfer chat**, while a voice one offers **End call** and **Transfer call** in their place.
+
+A dropped node lands with an auto-generated name, no prompt, and no edges. Its card says **No Prompt Specified** in red until you click it and write one. Connect it by dragging from the handle at the bottom of an existing node to the handle at the top of the new one, then set the condition on that edge.
+
+
+*Node types are dragged out of the palette, not clicked in*
+
+## Remove or copy a node
+
+Hover a node and two controls appear at its edge: a trash icon that removes it and a copy icon that duplicates it, prompt and all. Removing a node takes its edges with it, so the steps that fed into it are left with nowhere to go.
+
+That's worth knowing because **Save flow** validates before it writes: a graph with no start node, or with a step that connects to nothing, comes back with an error naming the problem. Close any gap you open, whether you got there by removing a node or by adding one you haven't wired up yet.
+
+The start node has no trash control, since a flow has to begin somewhere. To change where a conversation starts, edit that node rather than replacing it.
+
+## Dive deeper
+
+
+
+ Put more test cases through the flow you just edited
+
+
+ Give the simulator prompt something new to vary on
+
+
+ Put the scenario in front of your agent
+
+
diff --git a/src/pages/docs/simulation/guides/fix-my-agent.mdx b/src/pages/docs/simulation/guides/fix-my-agent.mdx
new file mode 100644
index 00000000..8e18ce96
--- /dev/null
+++ b/src/pages/docs/simulation/guides/fix-my-agent.mdx
@@ -0,0 +1,49 @@
+---
+title: "Fix My Agent"
+description: "Turn a finished run into a ranked list of issues and fixes, then hand off to an optimization"
+---
+
+**Fix My Agent** reads a run once it's finished and turns the calls in it into a short, ranked list of what's going wrong, each with a recommendation for what to change. You don't have to scroll through every transcript looking for a pattern yourself: it arrives already grouped, worst first, ready to hand off to an optimization once you've read through it.
+
+## Open it from a finished run
+
+On a run's [results page](/docs/simulation/guides/explore-results), **Fix My Agent** sits next to the tabs rather than inside one of them. Click it to open a side panel that stays open alongside whichever tab you're on, with a chevron to collapse it out of the way when you don't need it.
+
+The button only turns on once the run has enough to analyse: the run has to be completed, and it needs at least 15 connected calls behind it. A run still in progress, or one with fewer calls than that, leaves the button disabled, with a tooltip telling you which of the two is missing.
+
+## Generate the analysis
+
+The first time you open the panel on a run, it's empty: "There are no suggestions yet, click the refresh button to get suggestions." Click refresh to run the analysis over the run's calls. If it genuinely finds nothing worth flagging, it says so instead of manufacturing an issue to fill the space.
+
+## What a prioritised issue looks like
+
+Each entry in the list is one issue, not one call. A run where a dozen calls fail the same way for the same reason surfaces as a single entry, not a dozen. Every entry carries:
+
+- A short **heading** naming the issue
+- A **priority**, high, medium, or low, so you know which to read first
+- A written **recommendation** of what to change to address it
+- The **calls it's drawn from**. Click the entry and the calls grid on the page narrows to just those, so you can read the transcripts behind the pattern before you act on it
+
+The list carries two counts: how many issues were found in total, and how many of those are written as a fix to your agent's prompt, the ones you can act on directly. You can switch between issues at the agent level and issues at the domain level. A separate group covers issues that aren't about the prompt at all, things about how the agent is set up rather than what it says. Those are worth reading, but they're informational: an optimization run can't act on them the way it can on a prompt-based suggestion.
+
+## Hand off to an optimization
+
+Once you've read through the actionable suggestions, **Optimize My Agent** is the button that moves you from reading recommendations to acting on them automatically. It opens the optimization setup scoped to this run and the issues you were just looking at.
+
+[Running optimizations](/docs/simulation/guides/running-optimizations) walks through finishing that setup and starting the run. What the run does once it starts, searching for a better prompt and scoring each candidate against your [evals](/docs/evaluation), is covered in [Optimization](/docs/simulation/concepts/optimization).
+
+The optimization you start this way lands back on the same run's results page, under its **Optimization Runs** tab, alongside any others you've started from here.
+
+## Dive deeper
+
+
+
+ Finish the setup and start an optimization run
+
+
+ Read the trials and apply the winning prompt
+
+
+ What an optimization run searches for, and how
+
+
diff --git a/src/pages/docs/simulation/guides/optimization-runs.mdx b/src/pages/docs/simulation/guides/optimization-runs.mdx
new file mode 100644
index 00000000..8caabc37
--- /dev/null
+++ b/src/pages/docs/simulation/guides/optimization-runs.mdx
@@ -0,0 +1,55 @@
+---
+title: "Optimization runs"
+description: "Track an optimization run through its steps and trials, and apply the prompt that wins"
+---
+
+An **optimization run** is what you get after [Fix My Agent](/docs/simulation/guides/fix-my-agent) points the search at a finished [run](/docs/simulation/concepts/runs-and-results), covered in [Running optimizations](/docs/simulation/guides/running-optimizations). This page covers reading one you've already started: its steps, its trials, the score behind each trial, and what to do with the one that wins.
+
+ ST["Steps initializing, baseline, trials, finalizing"]
+ OR --> TR["Trials"]
+ TR --> BASE["Baseline trial your current prompt, unchanged"]
+ TR --> CAND["Candidate trials one prompt each"]
+ CAND --> BEST(["Best trial highest score, flagged for you"])
+`} />
+
+## Find a run's optimization history
+
+Open a run's results page and switch to the **Optimization Runs** tab, one of the three tabs covered in [Explore results](/docs/simulation/guides/explore-results). It lists every optimization attempt made against that execution, one row per attempt, with its name, how many trials it ran, which optimizer it used, and its status. A run with none yet shows **No optimization runs found**.
+
+Click a row to open that attempt.
+
+## Watch it move through its steps
+
+The header repeats the run's name and its status: pending, running, completed, or failed. Alongside it sit when the run started, which optimizer ran (Random Search, Bayesian, ProTeGi, Meta-Prompt, PromptWizard, or GEPA), and which model ran it. A **Parameters** button opens a popover listing the values you set when you created the run; its **Learn more** link goes to the [Optimization](/docs/optimization) product docs, which cover each optimizer's parameters in depth.
+
+Below the header, an **Optimization Steps** section tracks the run through four stages: setting up, scoring your current prompt as a baseline, running the search, and finalizing the result. It keeps itself current while the run is still going, so you can leave the page and come back to see how far it's got.
+
+Once a run is completed or failed, a **Rerun Optimization** button appears in the header. It opens a dialog prefilled with this run's optimizer, model, and parameters, all editable before you submit, and submitting creates a new optimization run rather than restarting this one.
+
+## Read the trials and the score per trial
+
+Every optimization run produces trials. The first is always the baseline: it scores your prompt exactly as it stands today, before the search changes anything, and gives every later trial a line to beat. Each trial after it is a candidate the search tried, and carries its own prompt text, its own average score, and how that score moved against the baseline.
+
+The best-performing trial, the one with the highest average score among everything the search actually tried, is flagged so you don't have to hunt for it. Open it to read its full prompt text, plus which [evals](/docs/evaluation) scored it and which [scenarios](/docs/simulation/concepts/scenarios) it ran against, the same ones your original run used.
+
+## Apply the winning configuration
+
+
+No button pushes a trial's prompt back onto your agent for you. Read the winning trial, copy its prompt text, and paste it into a new version of your agent from [Connect your agent](/docs/simulation/guides/connect-your-agent).
+
+
+Once that version exists, treat it like any other change: start a new [simulation](/docs/simulation/guides/create-simulation) against it and compare the results to the run the optimization started from, rather than trusting the trial's score on its own to carry over.
+
+## Dive deeper
+
+
+
+ Add the winning prompt as a new agent version
+
+
+ Start another optimization run
+
+
diff --git a/src/pages/docs/simulation/guides/prompt-simulation.mdx b/src/pages/docs/simulation/guides/prompt-simulation.mdx
new file mode 100644
index 00000000..bcfeb107
--- /dev/null
+++ b/src/pages/docs/simulation/guides/prompt-simulation.mdx
@@ -0,0 +1,55 @@
+---
+title: "Simulate a prompt"
+description: "Run a prompt version from the Prompt Workbench through chat scenarios, without an agent definition or any code."
+---
+
+A prompt simulation runs a saved prompt version through chat scenarios directly from the Prompt Workbench. There's no agent definition to create first and nothing to run on your side: the prompt version itself plays the assistant side of the conversation, and Future AGI drives both ends of the chat.
+
+This is the fastest way to see how one prompt version holds up over a multi-turn conversation before it's wired into an agent or shipped anywhere. If the prompt is meant to sit behind an agent instead, [Connect your agent](/docs/simulation/guides/connect-your-agent) and [Create a simulation](/docs/simulation/guides/create-simulation) cover that path; simulating the prompt directly skips both.
+
+
+Prompt versions themselves, drafts, labels, and how a template is saved, belong to the Prompt Workbench. [Versions and Labels](/docs/prompt/concepts/versions-and-labels) covers that; this page only covers running one through a simulation.
+
+
+## Open the Simulation tab
+
+In the Prompt Workbench, open the prompt template you want to test and switch to its **Simulation** tab.
+
+The tab stays disabled until the template has at least one saved, non-draft version with content in it. Land on it too early and it explains why instead of opening: "You need to submit at least one prompt before running simulations" before anything's been saved, or "Save your prompt to run simulations" once a draft exists but nothing's been submitted yet. Save a version first in either case.
+
+## Create the simulation
+
+Once the tab opens, its **Simulation Runs** header lists every simulation already run against this template. Start a new one and a **Create Chat Simulation** dialog opens with:
+
+- **Simulation Name**, required
+- **Prompt Version**, the saved version of this template that plays the assistant, shown with **Default** and **Draft** chips so you can tell which one you're picking
+- **Description**, optional
+- **Select Scenarios**, the chat scenarios already in the workspace, with **Select All** and **Deselect All** to move fast, plus a **Create New Chat Scenario** row if none fit yet. That row starts the same process as [Create scenarios](/docs/simulation/guides/create-scenarios), just from inside this dialog
+
+Submitting cycles the button through **Create Simulation**, **Creating...**, and **Starting...**, and the run begins on its own from there.
+
+## What it runs against
+
+Unlike a simulation built from an agent definition, this one has no agent and no deployment behind it. The prompt version is the thing under test, so Future AGI can call it directly and there's nothing to connect or drive from your own code. That's also why it's chat only: a prompt has no phone number or voice provider attached to it, so a prompt simulation always runs as text, whatever channel the eventual agent will use.
+
+The scenarios work exactly as they do anywhere else: each row plays one conversation, and its [persona](/docs/simulation/concepts/personas) drives the customer side. See [Scenarios](/docs/simulation/concepts/scenarios) for how one is put together.
+
+## How results differ from an agent simulation
+
+The [run](/docs/simulation/concepts/runs-and-results) it produces reads like any chat run: **Chat Details**, not **Call Details**, throughout, since text is the only mode it ever runs in. The same chat metrics apply, CSAT, token counts, latency, turn count. Voice-only metrics like talk ratio, interruption rate, and words per minute never apply, because a prompt simulation never places a call.
+
+One control an agent-based chat run has is missing here: the **Re-run simulation** button on the run's header doesn't appear for a prompt-sourced run at all. To test a change, create a new simulation instead, most often the same scenarios against a different prompt version, and compare the results of the two runs.
+
+## Dive deeper
+
+
+
+ Build the chat scenarios a prompt simulation plays against
+
+
+ Run the same kind of chat scenario against a full agent instead
+
+
+ Read the calls, transcripts, and metrics a run leaves behind
+
+
diff --git a/src/pages/docs/simulation/guides/replay-chat.mdx b/src/pages/docs/simulation/guides/replay-chat.mdx
new file mode 100644
index 00000000..6d7f21d7
--- /dev/null
+++ b/src/pages/docs/simulation/guides/replay-chat.mdx
@@ -0,0 +1,73 @@
+---
+title: "Replay chat sessions"
+description: "Turn a real production chat session into a scenario, then rerun it against your agent"
+---
+
+[Replay](/docs/simulation/concepts/replay) turns one production conversation into a scenario you can run in Simulation. This guide is the how-to half: finding the session, what replay carries over from it, and what you get back once you run it. For what replay is, and how session replay differs from trace replay, read the concept page first.
+
+
+Replay only sees conversations [Observe](/docs/observe/concepts/sessions) has already recorded, so your chat agent needs to be sending traces under a shared `session.id` before there's anything to replay. You'll also need an [API key pair](/docs/admin-settings/api-keys) to run the replayed scenario, the same one any chat simulation uses.
+
+
+## Find the session
+
+Open [Explore dashboard](/docs/observe/guides/explore-dashboard) and look for the conversation you actually want to reproduce: a chat that gave a wrong answer, lost the thread partway through, or escalated when it shouldn't have. Observe groups a conversation's turns under one `session.id`, and that grouping is exactly what replay reads, so a conversation split across several session IDs won't come back as a single scenario.
+
+## Start the replay
+
+From the session, start a replay. Future AGI reads its turns in order and turns them into a [scenario](/docs/simulation/concepts/scenarios): one row that plays out the same conversation, in the order it actually happened. It also creates an [agent definition](/docs/simulation/concepts/agent-definitions) to hold that scenario, since every scenario has to belong to one.
+
+The agent definition it creates is just a place for the scenario to live; you still point the run at whichever version of your own agent you want to test.
+
+## What carries over
+
+- **The conversation.** Every turn of the session, in order, becomes the scenario's script. Run it and the simulator works through the same exchange your production user had, not one it invents
+- **No voice-only detail.** Provider configuration, recordings, and call-level metrics belong to voice; a chat replay carries none of it, since chat calls never had it to begin with
+
+## Run it
+
+A replayed scenario is an ordinary scenario, so run it the same way as any chat run: [create a simulation](/docs/simulation/guides/create-simulation) against it, then drive it from your own code with the SDK.
+
+```bash
+pip install agent-simulate
+```
+
+```python
+import asyncio
+from fi.simulate import TestRunner, AgentInput
+
+async def customer_support_agent(input: AgentInput) -> str:
+ user_message = input.new_message["content"] if input.new_message else ""
+ return await my_agent.respond(user_message)
+
+async def main():
+ runner = TestRunner()
+ await runner.run_test(
+ run_test_name="Replaying chat_1001", # your run's name, exactly
+ agent_callback=customer_support_agent,
+ )
+
+asyncio.run(main())
+```
+
+[Run a chat simulation](/docs/simulation/guides/run-chat-simulation) covers the callback in full, including returning an `AgentResponse` when your agent's tool calls need to show up in the results.
+
+## What comes back
+
+The run produces an execution with the replayed conversation's transcript and metrics, same as any chat run, readable from [Calls & transcripts](/docs/simulation/guides/explore-results/calls-and-transcripts) the same way you'd read any other one.
+
+Once it's finished, open the call and use **Compare with baseline chat**. It lines the replayed conversation up against the original side by side, transcripts and eval scores together, so you can see exactly what your change moved instead of just that a number went up.
+
+## Dive deeper
+
+
+
+ Wire the SDK callback the replayed scenario runs through
+
+
+ Read the replayed conversation turn by turn
+
+
+ Rerun a production call on its original voice configuration
+
+
diff --git a/src/pages/docs/simulation/guides/replay-voice.mdx b/src/pages/docs/simulation/guides/replay-voice.mdx
new file mode 100644
index 00000000..9732754f
--- /dev/null
+++ b/src/pages/docs/simulation/guides/replay-voice.mdx
@@ -0,0 +1,42 @@
+---
+title: "Replay voice calls"
+description: "Turn one production call into a run against your dev agent, then compare the two"
+---
+
+This walks through replaying a single production voice call: finding it, seeing what the replay carries over from the original, and reading what the run hands back. For what replay actually reconstructs and why that matters, see [Replay](/docs/simulation/concepts/replay).
+
+
+Voice replay needs [voice observability](/docs/observe/concepts/voice-observability) already capturing the call in Observe, and it only reconstructs a call's configuration for Vapi. A Retell or Bland.ai call replays too, but you only get a transcript comparison back, not a config carried across. Replay covers why.
+
+
+## Find the call
+
+Open the call from [Explore sessions & users](/docs/observe/features/session) and start a replay from there, as a trace since one voice call is one trace. If you're after a whole multi-call conversation instead of a single one, that's a session replay, and it works the same way from that point on.
+
+## What carries over
+
+Starting the replay doesn't touch the original call. It creates a copy for you to work against: a voice [agent definition](/docs/simulation/concepts/agent-definitions) carrying the provider, model, and assistant configuration that call actually ran on, and a [scenario](/docs/simulation/concepts/scenarios) built from its transcript. Edit either freely. Nothing you change here reaches the agent that's still taking real calls.
+
+## Run it
+
+The agent and scenario land in an ordinary [simulation run](/docs/simulation/guides/create-simulation), and a voice run starts placing calls the moment it's created, same as any other. [Run a voice simulation](/docs/simulation/guides/run-voice-simulation) covers that wizard in full.
+
+To test a fix, edit the recreated agent definition, or point the run at a newer version, then place a fresh call rather than only rescoring the old one: open the run and rerun it with **Run test + Evals**.
+
+## What comes back
+
+Once the new call finishes, open it and click **Compare with baseline** to see it next to the original. You get both transcripts side by side, both recordings to play back, and the call metrics for each: duration, agent latency, talk ratio, words per minute, and interruption counts. That comparison is what tells you whether the change actually moved anything, not just that the new call completed.
+
+## Dive deeper
+
+
+
+ Walk through the run wizard a voice agent uses
+
+
+ Read a call's transcript and details outside a replay comparison
+
+
+ What each voice provider supports
+
+
diff --git a/src/pages/docs/simulation/guides/run-chat-simulation.mdx b/src/pages/docs/simulation/guides/run-chat-simulation.mdx
new file mode 100644
index 00000000..7dd8f732
--- /dev/null
+++ b/src/pages/docs/simulation/guides/run-chat-simulation.mdx
@@ -0,0 +1,132 @@
+---
+title: "Run a chat simulation"
+description: "Create a chat simulation in the UI, then answer it from your own code with the agent-simulate SDK"
+---
+
+A chat simulation plays every row of its scenarios against your agent, but a chat agent lives in your own code, somewhere Future AGI can't reach on its own. You write a callback that answers on your agent's behalf, and the SDK calls it once per turn and carries the reply back into the conversation. This guide creates the run test, then implements that callback with the `agent-simulate` package.
+
+
+You need a chat [agent definition](/docs/simulation/concepts/agent-definitions) and at least one chat [scenario](/docs/simulation/concepts/scenarios) to build the run test against, and an [API key pair](/docs/admin-settings/api-keys) to drive it from your code.
+
+
+## Create the chat simulation
+
+Build the run test in [Create a simulation](/docs/simulation/guides/create-simulation): name it, pick a chat agent definition and version, tick the chat scenarios you want it to face, and attach evals. The wizard is the same one voice run tests use; choosing a chat agent definition on the first step is what narrows the second step to chat scenarios.
+
+Click **Run Simulation** on the last step and nothing plays yet. Creating the run test only saves the bundle, since there's no agent on Future AGI's side to call. Its **Simulated runs** tab shows the install and run snippet you're about to use, already carrying the run test's exact name.
+
+## Write the agent callback
+
+Install the SDK:
+
+```bash
+pip install agent-simulate
+```
+
+Each turn, the SDK calls your callback with an `AgentInput` and expects a plain string or an `AgentResponse` back:
+
+- `new_message`: the message to answer this turn, shaped `{"role": ..., "content": ...}`
+- `messages`: the full conversation so far, including that message
+- `thread_id`: identifies which conversation this turn belongs to
+- `execution_id`: the [run](/docs/simulation/concepts/runs-and-results) this call is part of
+
+A callback is either a plain async function or a class extending **AgentWrapper**; both take an `AgentInput` and return `Union[str, AgentResponse]`. `my_agent` in the examples below stands in for your own agent code; swap it for whatever actually answers your users.
+
+
+
+ ```python
+ from typing import Union
+ from fi.simulate import AgentInput, AgentResponse
+
+ async def agent_callback(input: AgentInput) -> Union[str, AgentResponse]:
+ user_text = input.new_message["content"] if input.new_message else ""
+ return await my_agent.respond(user_text)
+ ```
+
+
+ ```python
+ from typing import Union
+ from fi.simulate import AgentWrapper, AgentInput, AgentResponse
+
+ class MyAgent(AgentWrapper):
+ async def call(self, input: AgentInput) -> Union[str, AgentResponse]:
+ user_text = input.new_message["content"] if input.new_message else ""
+ return await my_agent.respond(user_text)
+
+ # pass an instance: agent_callback=MyAgent()
+ ```
+
+
+
+Return a plain string when there's nothing else to report. Return an `AgentResponse` when your agent called tools this turn:
+
+- `content` (required): the reply text
+- `tool_calls`: the tools your agent invoked
+- `tool_responses`: what those tools returned, as `{"role": "tool", "tool_call_id": ..., "content": ...}` entries
+- `metadata`: anything else you want attached to the turn
+
+```python
+return AgentResponse(
+ content="Let me check that order for you.",
+ tool_calls=[
+ {"id": "call_1", "type": "function", "function": {"name": "lookup_order", "arguments": '{"order_id": "123"}'}}
+ ],
+ tool_responses=[
+ {"role": "tool", "tool_call_id": "call_1", "content": '{"status": "shipped"}'}
+ ],
+)
+```
+
+Scoring those tool calls is a separate opt-in step, covered in [Evaluate tool calls](/docs/simulation/guides/evaluate-tool-calls).
+
+
+A conversation ends on its own once the scenario reaches its end condition, or after 50 turns if it hasn't, whichever comes first. If your callback raises an exception, the SDK marks that call failed (or completed, if an earlier turn already succeeded) and reports a generic error to the dashboard rather than your exception's message. Log the exception yourself if you need to know what actually went wrong.
+
+
+## Run the simulation
+
+Create a **TestRunner** and point `run_test` at the run test:
+
+```python
+import asyncio
+from fi.simulate import TestRunner, AgentInput
+
+async def customer_support_agent(input: AgentInput) -> str:
+ user_message = input.new_message["content"] if input.new_message else ""
+ return await my_agent.respond(user_message)
+
+async def main():
+ runner = TestRunner()
+ await runner.run_test(
+ run_test_name="Simulating support-agent-chat", # exact match to the run test's name
+ agent_callback=customer_support_agent,
+ concurrency=5,
+ )
+ print("Simulation finished. View results in the dashboard.")
+
+asyncio.run(main())
+```
+
+`TestRunner()` reads `FI_API_KEY` and `FI_SECRET_KEY` from your environment; pass `api_key`/`secret_key` directly if you'd rather not use env vars, and `api_url` (or `FI_BASE_URL`) if you're pointing at a self-hosted deployment. Missing credentials don't fail immediately, they only log a warning, then fail once the SDK actually calls the backend, so check both keys first if a run test stays empty.
+
+`run_test` takes `run_test_name`, matched exactly to the run test you created, or `run_id` if you already have it, plus your `agent_callback`. `concurrency` sets how many scenario rows it plays at once.
+
+
+The object `run_test` returns doesn't carry your results, its `results` list is always empty. Read outcomes from the run test's **Simulated runs**, **Chat Details**, and **Analytics** tabs instead.
+
+
+Run the script and Future AGI plays each scenario row as a [persona](/docs/simulation/concepts/personas), turn by turn, against your callback, until every row has either finished or failed.
+
+## Dive deeper
+
+
+
+ Read the transcripts and scores this run test produces
+
+
+ Send a single chat back through your agent after a fix
+
+
+ Full field and method reference for agent-simulate
+
+
diff --git a/src/pages/docs/simulation/guides/run-voice-simulation.mdx b/src/pages/docs/simulation/guides/run-voice-simulation.mdx
new file mode 100644
index 00000000..3573cfa2
--- /dev/null
+++ b/src/pages/docs/simulation/guides/run-voice-simulation.mdx
@@ -0,0 +1,53 @@
+---
+title: "Run a voice simulation"
+description: "Pick the version and provider dialing the calls, then read how they landed"
+---
+
+A voice run test starts itself the moment it's created, and **Run New Simulation** fires another full execution of the same bundle, both already covered in Create a simulation. What's left is specific to voice: which provider is actually dialing, what a call looks like while it's in progress, where the finished conversations land, and **Re-run simulation**, a control that acts on a single execution rather than starting a new one.
+
+
+This picks up after a voice run test already exists, built in [Create a simulation](/docs/simulation/guides/create-simulation) with a voice [agent definition](/docs/simulation/concepts/agent-definitions) attached. That guide covers the wizard itself, picking scenarios and evals; nothing here repeats it.
+
+
+## The definition decides the provider
+
+The **Choose Agent definition** and **Choose version** fields in the wizard are where this gets locked in, and for a voice run they carry more weight than they do for chat: the provider that places every call, Vapi or Retell, lives on the agent definition itself, not on the version. Every version under one definition dials through the same provider, so switching providers means pointing the run test at a different definition, not a different version of this one. [Voice providers](/docs/simulation/reference/voice-providers) covers what each one supports.
+
+Both choices are fixed once the run test is created. Getting the wrong version or provider costs more here than it does in chat, since undoing it means real call minutes already spent, not just a rerun of code.
+
+## While the calls are placed
+
+Each row in **Call Details** starts as a placeholder reading "Call has not been picked up yet." Once dialing starts on that row it switches to "Call is in progress," and stays there until the call ends and the row fills in with duration, status, and a recording. Rows fill in as their calls finish, not in the order they were listed, so a run midway through shows some rows done and others still waiting.
+
+A single call can run up to 30 minutes before it's cut off, so a scenario with a handful of long, wandering conversations takes a while to finish even at a small row count.
+
+**Stop Running**, in the header, is available for as long as the run is actively placing calls. It asks for confirmation through a **Confirm Stop Runs** dialog, and it's the way to cut a bad batch short instead of waiting out every remaining call.
+
+## Where you land when it's done
+
+You're still on **Call Details**, only now every row is a finished call instead of a placeholder. Open one and you land on the call itself: the recording to play back, the transcript beside it, the evals that scored it, and a cost breakdown split across speech-to-text, language model, and text-to-speech usage. Voice calls also carry metrics no chat call has, like talk ratio, interruption counts, and words-per-minute on both sides of the conversation.
+
+**Analytics** is the tab for looking across every call in the run rather than one at a time, and it's the same tab any run test uses, not something specific to voice. [Explore results](/docs/simulation/guides/explore-results) walks through both Logs and Analytics, and [Call metrics](/docs/simulation/reference/call-metrics) defines every number a voice call produces.
+
+## Running it again
+
+**Re-run simulation**, in the header, acts on the calls that already exist rather than starting a fresh execution. It's disabled with a tooltip when the run has no completed calls to re-simulate yet. Clicking it opens a choice between two options, and voice is the one channel that gets both:
+
+- **Run Evals** rescores the existing calls against the run's current eval configs, recording and transcript untouched. Reach for this after changing an eval's mapping, when you want updated scores without spending call minutes again.
+- **Run test + Evals** dials fresh calls for the run and scores those. Use it when the agent itself changed and the old recordings and transcripts no longer represent what it does.
+
+Either way, a **Confirm Rerun Test** dialog asks you to confirm before anything starts.
+
+## Dive deeper
+
+
+
+ Read transcripts, recordings, and scores across a full run
+
+
+ Send one call's transcript back through the agent to compare against the original
+
+
+ What each voice provider needs from an agent definition
+
+
diff --git a/src/pages/docs/simulation/guides/running-optimizations.mdx b/src/pages/docs/simulation/guides/running-optimizations.mdx
new file mode 100644
index 00000000..adc80984
--- /dev/null
+++ b/src/pages/docs/simulation/guides/running-optimizations.mdx
@@ -0,0 +1,55 @@
+---
+title: "Running optimizations"
+description: "Open the optimization drawer from Fix My Agent, pick an algorithm, and start the run"
+---
+
+Starting an optimization run doesn't happen on its own page. It hangs off a finished simulation's results, launched from inside [Fix My Agent](/docs/simulation/guides/fix-my-agent), and this guide walks the drawer that opens from there: picking an algorithm, setting its fields, and starting the run. What each algorithm actually does, and what to do with the trials once they land, are covered elsewhere and linked as you go.
+
+
+Optimization launches from inside Fix My Agent, so its prerequisites apply first: the run has to be complete, and it needs at least 15 connected calls before **Fix My Agent** is even clickable.
+
+
+## Open the drawer
+
+On a run's results page, click **Fix My Agent** in the tab bar. Inside the panel, under **Prompt based suggestions**, click **Optimize My Agent**. That opens a dialog titled **Choose optimization type**, and this is the drawer the rest of this guide configures.
+
+## Pick the algorithm
+
+**Choose Optimizer** is the first field, a search-select listing every algorithm: Random Search, Bayesian, ProTeGi, Meta-Prompt, PromptWizard, and GEPA. What each one does and when to reach for it is covered on [Optimization](/docs/simulation/concepts/optimization). For how an algorithm works internally, the [Optimization](/docs/optimization) product docs are the deeper reference.
+
+## Configure it
+
+Two fields sit above the algorithm-specific ones, the same for every pick:
+
+- **Name**: required, whatever tells you what this attempt was about
+- **Language Model**: the model this optimization run uses
+
+Below them, the parameter fields change with the algorithm:
+
+| Optimizer | Parameters |
+|---|---|
+| Random Search | Number Variations |
+| Bayesian | Min examples, Max examples, No.of trials |
+| ProTeGi | Number of gradients, Errors per gradient, Prompts per gradient, Beam size, Number of Rounds |
+| PromptWizard | Mutated Rounds, Refined Iterations, Beam size |
+| GEPA | Max Metric Calls |
+| Meta-Prompt | Number of Rounds |
+
+Every algorithm ends on the same field, **Optimization Objective**, multiline: write in your own words what you want this run to fix or improve.
+
+## Start it
+
+Click **Start Optimizing your agent** at the bottom of the drawer. A toast confirms it, "Optimization Created Successfully," and the run appears in the execution's **Optimization Runs** tab right away, moving from **pending** to **running** as it works through its trials.
+
+[Optimization runs](/docs/simulation/guides/optimization-runs) picks up from here: reading the trials as they come in and applying the one that wins.
+
+## Dive deeper
+
+
+
+ Read the trials as they land and apply the winner
+
+
+ What each algorithm does and when to reach for it
+
+
diff --git a/src/pages/docs/simulation/index.mdx b/src/pages/docs/simulation/index.mdx
index 1ff0b441..6fa106cb 100644
--- a/src/pages/docs/simulation/index.mdx
+++ b/src/pages/docs/simulation/index.mdx
@@ -1,41 +1,48 @@
---
-title: "Future AGI Simulation: Test Agents Before Production"
-description: "Test AI agents and prompts through controlled simulations before deploying to production. Run voice and chat simulations, score results, and iterate."
+title: "Overview"
+description: "Rehearse your agent on hard conversations, score each one, and fix what fails"
---
-## About
+Simulation runs your agent through realistic conversations before real customers ever reach it. You assemble a test from three pieces, an [agent definition](/docs/simulation/concepts/agent-definitions), a [scenario](/docs/simulation/concepts/scenarios), and a [persona](/docs/simulation/concepts/personas), run it as voice or chat, and score every conversation with evals you attach to the run. When a run turns up a failure, you fix the agent and run it again.
-Simulation lets you test voice and chat agents against simulated customers before going live. You define **agent definitions** (how to connect to your agent), **scenarios** (what the customer wants), and **personas** (who the customer is). The platform runs the conversations, records transcripts, and scores them with evaluations.
+You drive all of this from the dashboard. Chat simulations can also run from your own code with the [SDK](/docs/simulation/reference/sdk-api), and prompt versions can be simulated straight from [Prompt Workbench](/docs/prompt) with no deployed agent at all.
-
+## Catch failures before customers do
-## How Simulation Connects to Other Features
+A production incident is expensive to learn from. Simulation moves that learning earlier: the refund your agent botches or the caller it talks over shows up in a test run, not in front of a customer.
-- **Evaluation**: Scores every simulated conversation automatically. [Learn more](/docs/evaluation)
-- **Observability**: Simulation traces flow into Observe so you can replay and debug conversations. [Learn more](/docs/observe)
-- **Optimization**: Use simulation results to improve prompts with Fix My Agent. [Learn more](/docs/optimization)
-- **Datasets**: Create scenarios from datasets, and export results back for further analysis. [Learn more](/docs/dataset)
+Every run is inspectable. Each conversation comes back with:
-## Getting Started
+- the full transcript, and the audio recording for voice
+- conversation metrics like latency, interruptions, and talk ratio
+- an eval score per conversation, so a failure is something you open and read rather than guess at
-
-
- Run your agent against scenarios from the platform.
-
-
- Run chat simulations programmatically.
+## The agent development loop
+
+Simulation is the rehearsal stage of the agent development lifecycle: every change to your agent passes through it before production, and production feeds the next rehearsal.
+
+
+
+The loop closes on itself twice: a failing score sends you back to fix and re-simulate, and a production trace you [replay](/docs/simulation/concepts/replay) becomes a new test case.
+
+## How it connects
+
+- [Evaluation](/docs/evaluation) provides the templates that score each conversation
+- [Observe](/docs/observe) is the production counterpart: its traces flow back in through replay
+- [Optimization](/docs/optimization) improves the agent's prompt automatically from the results
+- [Datasets](/docs/dataset) seed scenarios in bulk and take results back for analysis
+- [Prompt Workbench](/docs/prompt) runs its prompt versions through the same simulations
+
+## Start here
+
+
+
+ How the pieces fit together before you run one
-
- Test prompts in multi-turn conversations.
+
+ Dial your agent and score the calls, step by step
-
- Get AI-powered diagnostics and fixes.
+
+ Drive your chat agent from the UI or the SDK
diff --git a/src/pages/docs/simulation/reference/built-in-personas.mdx b/src/pages/docs/simulation/reference/built-in-personas.mdx
new file mode 100644
index 00000000..49a3dd1d
--- /dev/null
+++ b/src/pages/docs/simulation/reference/built-in-personas.mdx
@@ -0,0 +1,165 @@
+---
+title: "Built-in personas"
+description: "The 18 built-in personas and every persona field, voice and chat, with the values each accepts"
+---
+
+Future AGI ships 18 built-in [personas](/docs/simulation/concepts/personas) your [scenarios](/docs/simulation/concepts/scenarios) can draw on, each with a fixed name, description, and trait set. This page lists all 18 with their exact values, plus every field a persona can hold, split into voice fields and chat fields. To build your own, see [Create personas](/docs/simulation/guides/create-personas).
+
+## The 18 built-in personas
+
+| Name | Description |
+|---|---|
+| The Impatient Driver | A truck driver who frequently uses a fuel app and gets frustrated when responses are slow or repetitive |
+| The Lost Newbie | A new truck driver exploring the fuel app for the first time who often asks for repeated guidance |
+| The Stressed Accountant | An overworked accountant managing multiple clients who remains polite but gets anxious about delays |
+| The Frustrated Subscriber | A business owner upset with repeated subscription billing issues despite being a long-time user |
+| The Confused First-Time User | A friendly teacher who recently joined the platform and needs reassurance while activating her account |
+| The Curious Evaluator | A manager evaluating a product for enterprise rollout, asking detailed and structured questions |
+| The No-Nonsense Executive | A confident female business owner who prefers concise, professional communication and fast decisions |
+| The Frustrated Everyday User | An emotional and impatient customer service professional expressing irritation casually yet openly |
+| The Reserved Senior | A retired senior who is cautious and skeptical, preferring calm, clear explanations |
+| The Emotional Loyalist | A marketing professional who's disappointed about recent changes but remains emotionally loyal |
+| The Hustling Homemaker | A motivated homemaker from India who manages family and side projects, looking for efficiency |
+| The Telecom Customer in Distress | An emotional and talkative young customer service worker irritated with telecom issues |
+| The Tech-Savvy Young Professional | A confident engineer who values efficiency and clear, technical communication |
+| The Polite Senior Caller | A retired Australian who is polite and friendly, seeking help with patience and courtesy |
+| The Hungry Customer in a Rush | A young female engineer frustrated by food delivery delays, switching between politeness and irritation |
+| The Local Restaurant Owner | A professional business owner concerned about operational details and customer experience |
+| The Delivery Driver on the Move | A talkative freelancer multitasking during deliveries, occasionally distracted while explaining issues |
+| The Enterprise IT Admin | A focused and analytical engineer managing system reliability and technical escalations |
+
+
+ All 18 built-in personas are **voice** personas. There are no built-in chat personas today; [running a chat simulation](/docs/simulation/guides/run-chat-simulation) draws only on custom personas you create yourself.
+
+
+## Every trait, per persona
+
+The traits below are split across three tables so each one stays readable, and every table is keyed on the persona name. All 18 personas appear in all three.
+
+### Who they are
+
+Demographics, as set on each built-in persona.
+
+| Name | Gender | Age | Location | Profession |
+|---|---|---|---|---|
+| The Impatient Driver | male | 32-40 | United States | Freelancer |
+| The Lost Newbie | male | 25-32 | United States | Freelancer |
+| The Stressed Accountant | male | 32-40 | Canada | Accountant |
+| The Frustrated Subscriber | male | 32-40 | United States | Business Owner |
+| The Confused First-Time User | female | 40-50 | United States | Teacher |
+| The Curious Evaluator | male | 32-40 | United Kingdom | Manager |
+| The No-Nonsense Executive | female | 40-50 | United States | Business Owner |
+| The Frustrated Everyday User | male | 32-40 | India | Customer Service |
+| The Reserved Senior | male | 60+ | United States | Retired |
+| The Emotional Loyalist | female | 32-40 | Australia | Marketing Professional |
+| The Hustling Homemaker | female | 32-40 | India | Homemaker |
+| The Telecom Customer in Distress | female | 25-32 | United States | Customer Service |
+| The Tech-Savvy Young Professional | male | 25-32 | South Africa | Engineer |
+| The Polite Senior Caller | male | 60+ | Australia | Retired |
+| The Hungry Customer in a Rush | female | 25-32 | United States | Engineer |
+| The Local Restaurant Owner | male | 40-50 | United States | Business Owner |
+| The Delivery Driver on the Move | male | 25-32 | United States | Freelancer |
+| The Enterprise IT Admin | male | 32-40 | United States | Engineer |
+
+
+### How they communicate
+
+Personality and speech traits.
+
+| Name | Personality | Communication style | Language | Accent |
+|---|---|---|---|---|
+| The Impatient Driver | Impatient and direct | Assertive | English | American |
+| The Lost Newbie | Friendly and cooperative | Questioning | English | American |
+| The Stressed Accountant | Detail-oriented | Technical | English | Canadian |
+| The Frustrated Subscriber | Impatient and direct | Direct and concise | English | American |
+| The Confused First-Time User | Friendly and cooperative | Questioning | English | Neutral |
+| The Curious Evaluator | Analytical | Detailed and elaborate | English | British |
+| The No-Nonsense Executive | Professional and formal | Direct and concise | English | American |
+| The Frustrated Everyday User | Emotional | Casual and friendly | English | Indian |
+| The Reserved Senior | Cautious and skeptical | Simple and clear | English | American |
+| The Emotional Loyalist | Emotional | Detailed and elaborate | English | Australian |
+| The Hustling Homemaker | Friendly and cooperative | Simple and clear | Hindi | Indian |
+| The Telecom Customer in Distress | Talkative | Casual and friendly | English | American |
+| The Tech-Savvy Young Professional | Confident | Technical | English | Neutral |
+| The Polite Senior Caller | Friendly and cooperative | Formal and polite | English | Australian |
+| The Hungry Customer in a Rush | Impatient and direct | Direct and concise | English | American |
+| The Local Restaurant Owner | Detail-oriented | Detailed and elaborate | English | American |
+| The Delivery Driver on the Move | Easy-going | Casual and friendly | English | Neutral |
+| The Enterprise IT Admin | Analytical | Technical | English | Neutral |
+
+
+### Voice behaviour
+
+Speed runs from 0.5 to 1.5, and both sensitivities run from 1 to 10. See [Voice fields](#voice-fields) for what each value means.
+
+| Name | Conversation speed | Background noise | Finished-speaking sensitivity | Interrupt sensitivity |
+|---|---|---|---|---|
+| The Impatient Driver | 1.25 | Yes | 6 | 6 |
+| The Lost Newbie | 1.0 | Yes | 5 | 5 |
+| The Stressed Accountant | 1.0 | No | 5 | 6 |
+| The Frustrated Subscriber | 1.25 | Yes | 6 | 6 |
+| The Confused First-Time User | 0.75 | No | 5 | 4 |
+| The Curious Evaluator | 1.0 | No | 6 | 5 |
+| The No-Nonsense Executive | 1.5 | No | 7 | 7 |
+| The Frustrated Everyday User | 1.25 | Yes | 6 | 6 |
+| The Reserved Senior | 0.75 | No | 4 | 3 |
+| The Emotional Loyalist | 1.0 | No | 5 | 5 |
+| The Hustling Homemaker | 1.25 | Yes | 5 | 5 |
+| The Telecom Customer in Distress | 1.25 | Yes | 5 | 6 |
+| The Tech-Savvy Young Professional | 1.25 | No | 6 | 6 |
+| The Polite Senior Caller | 0.75 | No | 4 | 3 |
+| The Hungry Customer in a Rush | 1.5 | Yes | 6 | 7 |
+| The Local Restaurant Owner | 1.0 | Yes | 5 | 6 |
+| The Delivery Driver on the Move | 1.25 | Yes | 4 | 4 |
+| The Enterprise IT Admin | 1.25 | No | 7 | 7 |
+
+
+ Two built-in personas carry values outside the standard field lists below: **The Curious Evaluator** uses accent `British`, which isn't in the accent picker, and **The Tech-Savvy Young Professional** is based in `South Africa`, which isn't one of the five standard locations. Both still work in a run; a filter or search built against the standard value lists just won't match them.
+
+
+## Persona fields
+
+A persona is built from a fixed set of fields. The type you pick at creation, voice or chat, decides which set applies; the fields below cover both.
+
+### Common fields
+
+| Field | Allowed values |
+|---|---|
+| Name | Free text, required |
+| Description | Free text, required |
+| Gender | `male`, `female` |
+| Age | `18-25`, `25-32`, `32-40`, `40-50`, `50-60`, `60+` |
+| Location | `United States`, `Canada`, `United Kingdom`, `Australia`, `India` |
+| Profession | `Student`, `Teacher`, `Engineer`, `Doctor`, `Nurse`, `Business Owner`, `Manager`, `Sales Representative`, `Customer Service`, `Technician`, `Consultant`, `Accountant`, `Marketing Professional`, `Retired`, `Homemaker`, `Freelancer`, `Truck Driver`, `Other` |
+| Personality | `Friendly and cooperative`, `Professional and formal`, `Cautious and skeptical`, `Impatient and direct`, `Detail-oriented`, `Easy-going`, `Anxious`, `Confident`, `Analytical`, `Emotional`, `Reserved`, `Talkative` |
+| Communication style | `Direct and concise`, `Detailed and elaborate`, `Casual and friendly`, `Formal and polite`, `Technical`, `Simple and clear`, `Questioning`, `Assertive`, `Passive`, `Collaborative` |
+| Language | `Arabic`, `Bengali`, `Bulgarian`, `Chinese`, `Croatian`, `Czech`, `Danish`, `Dutch`, `English`, `Filipino`, `Finnish`, `French`, `Georgian`, `German`, `Greek`, `Gujarati`, `Hebrew`, `Hindi`, `Hungarian`, `Indonesian`, `Italian`, `Japanese`, `Kannada`, `Korean`, `Malay`, `Malayalam`, `Mandarin`, `Marathi`, `Norwegian`, `Polish`, `Portuguese`, `Punjabi`, `Romanian`, `Russian`, `Slovak`, `Spanish`, `Swedish`, `Tagalog`, `Tamil`, `Telugu`, `Thai`, `Turkish`, `Ukrainian`, `Vietnamese` |
+| Multilingual | `true`, `false` |
+| Custom properties | Key-value pairs you name yourself |
+| Additional instructions | Free text |
+
+### Voice fields
+
+| Field | Allowed values |
+|---|---|
+| Accent | `American`, `Arabic`, `Australian`, `Bengali`, `Brazilian`, `Bulgarian`, `Canadian`, `Chinese`, `Croatian`, `Czech`, `Danish`, `Dutch`, `Filipino`, `Finnish`, `French`, `Georgian`, `German`, `Greek`, `Gujarati`, `Hebrew`, `Hungarian`, `Indian`, `Indonesian`, `Italian`, `Japanese`, `Kannada`, `Korean`, `Malay`, `Malayalam`, `Malaysian`, `Mandarin`, `Marathi`, `Neutral`, `Norwegian`, `Polish`, `Portuguese`, `Punjabi`, `Romanian`, `Russian`, `Slovak`, `South American`, `Southern`, `Spanish`, `Swedish`, `Tagalog`, `Tamil`, `Telugu`, `Thai`, `Turkish`, `Ukrainian`, `Vietnamese` |
+| Conversation speed | `0.5` (very slow), `0.75` (slow), `1.0` (moderate), `1.25` (fast), `1.5` (very fast) |
+| Background noise | `true`, `false`, set with a toggle in the form |
+| Finished-speaking sensitivity | `1` to `10` |
+| Interrupt sensitivity | `1` to `10` |
+
+
+ Finished-speaking sensitivity and interrupt sensitivity run in opposite directions of feel: a low finished-speaking sensitivity waits longer before assuming your agent is done talking, while a low interrupt sensitivity means the persona barely reacts to being talked over.
+
+
+### Chat fields
+
+| Field | Allowed values |
+|---|---|
+| Tone | `formal`, `neutral`, `casual` |
+| Verbosity | `brief`, `balanced`, `detailed` |
+| Regional Mix | `none`, `light`, `moderate`, `heavy` |
+| Slang Level | `none`, `light`, `moderate`, `heavy` |
+| Typo Level | `none`, `rare`, `occasional`, `frequent` |
+| Punctuation Style | `clean`, `minimal`, `expressive`, `erratic` |
+| Emoji Frequency | `never`, `light`, `regular`, `heavy` |
diff --git a/src/pages/docs/simulation/reference/call-metrics.mdx b/src/pages/docs/simulation/reference/call-metrics.mdx
new file mode 100644
index 00000000..c49b09dd
--- /dev/null
+++ b/src/pages/docs/simulation/reference/call-metrics.mdx
@@ -0,0 +1,95 @@
+---
+title: "Call metrics"
+description: "Field-by-field reference for call metrics and transcript speaker roles"
+---
+
+Every completed [call](/docs/simulation/concepts/runs-and-results) computes a set of metrics from its transcript and recording. This page lists each one by its field name, what it measures, and which channel (voice, chat, or both) it applies to, plus the transcript's speaker roles. For how these numbers roll up across a whole run, see [Analytics & metrics](/docs/simulation/guides/explore-results/analytics); for reading one call's transcript in the UI, see [Calls & transcripts](/docs/simulation/guides/explore-results/calls-and-transcripts).
+
+## Overall score
+
+| Field | Channel | Meaning |
+|---|---|---|
+| `overall_score` | Voice | A CSAT (customer satisfaction) score from 1 to 10, evaluated from the call recording. If that evaluation can't be parsed, the call falls back to a pass/fail signal reported by the voice provider instead |
+| `overall_score` | Chat | The same field, but computed as a CSAT score from the transcript directly, since there's no recording to evaluate |
+
+CSAT is the label you'll see on this field wherever it's surfaced in the product.
+
+## Talk time and interruptions
+
+Voice-only. Computed from how much of the call each party spent speaking.
+
+| Field | Unit | Meaning |
+|---|---|---|
+| `talk_ratio` | ratio | How much of the call the agent spent talking versus the caller. Shown as a split between agent talk percentage and customer talk percentage |
+| `user_interruption_count` | count | How many times the caller interrupted the agent |
+| `user_interruption_rate` | rate | How often the caller interrupted, as interruptions per call |
+| `ai_interruption_count` | count | How many times the agent interrupted the caller |
+| `ai_interruption_rate` | rate | How often the agent interrupted, as interruptions per call |
+| `avg_stop_time_after_interruption_ms` | ms | Average time the interrupted party takes to stop talking once interrupted |
+
+## Speaking pace
+
+Voice-only.
+
+| Field | Unit | Meaning |
+|---|---|---|
+| `user_wpm` | words/min | The caller's speaking pace |
+| `bot_wpm` | words/min | The agent's speaking pace |
+
+## Latency and response time
+
+| Field | Channel | Unit | Meaning |
+|---|---|---|---|
+| `avg_agent_latency_ms` | Voice | ms | Average time the agent takes to respond after the caller stops talking |
+| `avg_latency_ms` | Chat | ms | Average time the agent takes to respond after the user's message |
+| `response_time_ms` | Voice | ms | Average duration of the agent's own turns, computed from the transcript. This measures how long the agent's replies run, not how quickly it starts them |
+
+## Duration
+
+| Field | Unit | Meaning |
+|---|---|---|
+| `duration_seconds` | seconds | Length of the call. Voice calls are capped at 1800 seconds (30 minutes) |
+
+A run's total duration is the sum of its calls' `duration_seconds`, not a separately measured value.
+
+## Turn count and token usage
+
+Chat-only.
+
+| Field | Unit | Meaning |
+|---|---|---|
+| `turn_count` | count | Number of back-and-forth exchanges in the conversation |
+| `total_tokens` | tokens | Total tokens consumed by the agent's language model calls during the chat |
+| `input_tokens` | tokens | Tokens sent to the model as input |
+| `output_tokens` | tokens | Tokens generated by the model as output |
+
+## Cost breakdown
+
+| Field | Meaning |
+|---|---|
+| `cost_cents` | Total cost of the call, in cents |
+| `stt_cost_cents` | Cost of converting the caller's speech to text |
+| `llm_cost_cents` | Cost of the language model calls that drove the conversation |
+| `tts_cost_cents` | Cost of converting the agent's replies to speech |
+| `storage_cost_cents` | Cost of storing the call recording |
+
+
+ `stt_cost_cents` and `tts_cost_cents` only apply to voice calls: there's no audio to convert on chat, so these fields are never populated there.
+
+
+## Transcript and speaker roles
+
+Each turn in a call's transcript carries its text content, a start and end timestamp in milliseconds, and, for voice calls, a confidence score from speech recognition.
+
+| Speaker role | Shown in the transcript view | Meaning |
+|---|---|---|
+| `USER` | Yes | The caller or chat user's turn |
+| `ASSISTANT` | Yes | The agent's turn |
+| `SYSTEM` | Yes | A system-level turn in the conversation |
+| `TOOL_CALLS` | No | Records which tool the agent invoked. Feeds [tool-call evaluation](/docs/simulation/guides/evaluate-tool-calls) rather than the visible transcript |
+| `TOOL_CALL_RESULT` | No | The result a tool call returned. Also feeds tool-call evaluation, not the visible transcript |
+| `UNKNOWN` | No | A turn that didn't match any of the above; rare |
+
+
+ `TOOL_CALLS` and `TOOL_CALL_RESULT` turns only get written on voice calls, and tool-call evaluation is only available for agents on Vapi.
+
diff --git a/src/pages/docs/simulation/reference/sdk-api.mdx b/src/pages/docs/simulation/reference/sdk-api.mdx
new file mode 100644
index 00000000..1ebad4b4
--- /dev/null
+++ b/src/pages/docs/simulation/reference/sdk-api.mdx
@@ -0,0 +1,97 @@
+---
+title: "SDK & API"
+description: "Field-level reference for the SDK's callback contract and REST endpoints."
+---
+
+This page is the reference for reaching Simulation from your own code: the `agent-simulate` Python package's callback contract and `TestRunner`, and the REST endpoints it calls to execute a run. It covers chat agents only, since a [chat agent](/docs/simulation/concepts/agent-definitions) is answered by your own code rather than dialed over the phone. If you haven't connected one yet, start with [Connect your agent](/docs/simulation/guides/connect-your-agent). For the guided walkthrough, see [Run a chat simulation](/docs/simulation/guides/run-chat-simulation).
+
+## Install and authenticate
+
+```bash
+pip install agent-simulate
+```
+
+```python
+from fi.simulate import TestRunner, AgentInput, AgentResponse, AgentWrapper
+```
+
+`TestRunner` takes `api_key`, `secret_key`, and `api_url`, each falling back to an environment variable if omitted:
+
+| Argument | Env var | Default |
+|---|---|---|
+| `api_key` | `FI_API_KEY` | none |
+| `secret_key` | `FI_SECRET_KEY` | none |
+| `api_url` | `FI_BASE_URL` | `https://api.futureagi.com` |
+
+Requests carry the key and secret as `x-api-key` and `x-secret-key` headers. A missing key or secret only logs a warning at construction time; nothing fails until the first request comes back `401`.
+
+## The callback contract
+
+Your code is the agent. Each turn, the SDK calls your `agent_callback` with an `AgentInput` and expects back a `str` or an `AgentResponse`.
+
+**`AgentInput`**
+
+| Field | Type | Required | Description |
+|---|---|---|---|
+| `thread_id` | `str` | yes | Identifies the conversation this turn belongs to |
+| `messages` | `List[Dict[str, str]]` | yes | The full conversation so far, including the latest simulator message |
+| `new_message` | `Optional[Dict[str, str]]` | no | The latest simulator message, the one to reply to this turn |
+| `execution_id` | `Optional[str]` | no | Correlates this turn back to the run, for your own logging |
+
+**`AgentResponse`**
+
+| Field | Type | Required | Description |
+|---|---|---|---|
+| `content` | `str` | yes | The reply text sent back to the simulator |
+| `tool_calls` | `Optional[List[Dict[str, Any]]]` | no | Tool calls your agent made this turn |
+| `tool_responses` | `Optional[List[Dict[str, Any]]]` | no | Results for those tool calls, each a dict with `role`, `tool_call_id`, `content` |
+| `metadata` | `Optional[Dict[str, Any]]` | no | Free-form extra data; also accepts `metadata["tool_outputs"]` as `{"call_id": ..., "output": ...}` entries, an alternate way to report tool results |
+
+Returning a bare `str` is shorthand for `AgentResponse(content=...)` with everything else empty:
+
+```python
+async def agent_callback(input: AgentInput) -> str:
+ user_text = (input.new_message or {}).get("content", "") or ""
+ return f"Echo: {user_text}"
+```
+
+`AgentWrapper` is the class form: an abstract base class where you implement `async def call(self, input: AgentInput) -> Union[str, AgentResponse]` and pass an instance as `agent_callback` instead of a function. Either shape works: a plain `async def` function is wrapped automatically.
+
+
+If `call()` raises, the SDK doesn't forward your exception. It reports a generic error to the platform and marks the call `completed` if at least one earlier turn already succeeded, or `failed` if it fails on the first turn. Log the real error on your own side; it won't show up in the transcript.
+
+
+The conversation ends when the platform reports the chat as ended, or after 50 turns, whichever comes first.
+
+## TestRunner.run_test
+
+```python
+runner = TestRunner() # reads FI_API_KEY / FI_SECRET_KEY / FI_BASE_URL from env
+report = await runner.run_test(
+ run_test_name="Chat test", # or run_id=""
+ agent_callback=agent_callback,
+ concurrency=1,
+)
+```
+
+- Exactly one of `run_id` or `run_test_name` identifies which [run test](/docs/simulation/concepts/runs-and-results) to execute; `run_test_name` must match the simulation's name exactly, the same one it's created under in the UI or via [Create a simulation](/docs/simulation/guides/create-simulation)
+- `agent_callback` is your callback function or `AgentWrapper` instance
+- `concurrency` controls how many calls run in parallel
+
+
+`run_test` returns a `TestReport`, but in the current release its `results` field is always empty. Transcripts, metrics, and evaluations live on the platform, not on the returned object; read them from the dashboard or the REST endpoints below.
+
+
+## REST endpoints
+
+These are the endpoints `agent-simulate` calls on your behalf while `run_test` runs. They're useful for building your own client outside the SDK, or for understanding what a run does over the network. Authentication is the same `x-api-key` / `x-secret-key` headers as above.
+
+| Method | Path | Purpose |
+|---|---|---|
+| `GET` | `/simulate/run-tests/get-id-by-name/{run_test_name}/` | Resolve a run test's ID from its exact name |
+| `POST` | `/simulate/run-tests/{run_test_id}/chat-execute/` | Start a chat execution for a run test |
+| `POST` | `/simulate/test-executions/{test_execution_id}/chat/call-executions/batch/` | Create a batch of call executions under a test execution |
+| `POST` | `/simulate/call-executions/{call_execution_id}/chat/send-message/` | Send one turn's message on a call execution |
+| `PATCH` | `/simulate/call-executions/{call_execution_id}/` | Update a call execution's status |
+
+Paths are relative to the same base URL as the SDK, `https://api.futureagi.com` unless `FI_BASE_URL` overrides it.
diff --git a/src/pages/docs/simulation/reference/voice-providers.mdx b/src/pages/docs/simulation/reference/voice-providers.mdx
new file mode 100644
index 00000000..217bfcc1
--- /dev/null
+++ b/src/pages/docs/simulation/reference/voice-providers.mdx
@@ -0,0 +1,48 @@
+---
+title: "Voice providers"
+description: "Credentials and supported features for the voice providers Simulation connects to"
+---
+
+Simulation places voice calls through a connected provider. Vapi, Retell, and Bland.ai are the three native providers, chosen when you set up a voice [agent definition](/docs/simulation/concepts/agent-definitions). **Others** covers any agent you can reach by phone.
+
+## Vapi
+
+Connecting a voice agent to Vapi needs the API key and the assistant ID from your Vapi account, both entered when you [connect your agent](/docs/simulation/guides/connect-your-agent). If the agent takes inbound calls, it also needs a contact number of 10 to 12 digits.
+
+## Retell
+
+Connecting a voice agent to Retell needs the API key and the assistant ID from your Retell account, entered the same way as Vapi. Inbound calling needs the same 10 to 12 digit contact number.
+
+
+Retell agents don't support tool call evaluation. The **"Enable tool call evaluation"** toggle on a run has no effect on Retell-backed calls; evaluation covers the conversation only.
+
+
+## Bland.ai
+
+Bland.ai appears as **Bland.ai** in the provider dropdown and authenticates with an API key, the same as Vapi and Retell. Two things work differently:
+
+- **The Assistant ID field takes a Conversational Pathway ID.** Bland has no separate assistant object, so open the pathway your agent runs and copy its ID into Assistant ID
+- **A contact number is always required**, not just for inbound. Bland has no web connector, so every simulated call is placed over the phone. For inbound tests, use a Bland number attached to that pathway
+
+
+Bland records a single combined audio track rather than separate caller and agent channels. An eval whose input is mapped to the stereo, assistant, or customer recording resolves empty on a Bland call, so map whole-conversation evals to `call.voice_recording` instead. See [Edit a run's evals](/docs/simulation/guides/edit-evals) for where that mapping lives.
+
+
+Because Bland's API key is sent in a different header format from the other providers, switching an existing agent definition away from Bland clears the stored key. Re-enter it for the new provider.
+
+## What each provider supports
+
+| Capability | Vapi | Retell | Bland.ai |
+|---|---|---|---|
+| Voice calls | Yes | Yes | Yes |
+| Call recordings | Yes | Yes | Yes, one combined track |
+| Call transcripts | Yes | Yes | Yes |
+| Compare with baseline (replay) | Yes, the original call's configuration comes across | Transcript only | Transcript only |
+| Tool call evaluation | Yes | No | No |
+| Contact number | Inbound only | Inbound only | Always required |
+
+Recordings and transcripts show up in the [call detail view](/docs/simulation/guides/explore-results/calls-and-transcripts) the same way regardless of provider. All three support **"Compare with baseline"** on a call, which lines its transcript up against the reference call. What differs is how much of the original comes across, and [Replay](/docs/simulation/concepts/replay) is where that difference is explained.
+
+## Limits
+
+Voice calls are capped at 30 minutes on any provider; a call in progress is ended automatically once it hits that mark. See [Call metrics](/docs/simulation/reference/call-metrics) for how duration, cost and talk time are reported once a call completes.
diff --git a/src/pages/docs/simulation/troubleshooting.mdx b/src/pages/docs/simulation/troubleshooting.mdx
new file mode 100644
index 00000000..4f7adb3c
--- /dev/null
+++ b/src/pages/docs/simulation/troubleshooting.mdx
@@ -0,0 +1,167 @@
+---
+title: "Simulation FAQ & fixes"
+description: "Common simulation questions, and fixes for the errors you hit most"
+---
+
+## In this page
+
+The questions people ask most about Simulation, and the errors they run into, with a direct fix for each. Hit an error? Jump straight to [Common errors and fixes](#common-errors-and-fixes). If your answer isn't here, reach out via [support](https://futureagi.com/contact-us).
+
+## Common errors and fixes
+
+| Symptom | Cause | Fix |
+|---|---|---|
+| Scenario generation comes back **Failed** | An uploaded script or SOP is a scanned PDF with no text layer, or an imported dataset is under 10 rows, has duplicate column names, or has a `persona` column typed as plain text | Fix the source material: scripts and SOPs need a real text layer, datasets need at least 10 rows, unique column names, and a `persona` column typed as Persona if one exists |
+| A scenario is greyed out in the run wizard, tooltip "This scenario has no datapoints to run against" | The scenario has 0 rows | Open it and [add rows](/docs/simulation/guides/explore-scenarios/add-rows) before selecting it in a run |
+| "No.of rows" or "No. of scenarios" rejects the number you typed | Row count has to be between 10 and 20,000 | Enter a value in that range |
+| The **Add Column** drawer won't save more than 10 fields in one pass | Ten columns is the limit per save | Save those 10, then reopen **Add Column** for more; there's no cap on the scenario as a whole |
+| **Re-run simulation** is disabled, or missing entirely | The run has zero completed calls yet, or it's [simulated from a Prompt Workbench prompt](/docs/simulation/guides/prompt-simulation), which never gets a rerun | Wait for at least one call to finish; a prompt-sourced run needs a new simulation instead |
+| Removing an eval from a run is blocked | It's the last eval left on that run | [Add a replacement eval](/docs/simulation/guides/edit-evals) first; a run always needs at least one |
+| **Fix My Agent** stays disabled | The run isn't complete yet, or it has fewer than 15 connected calls | Wait for the run to finish with 15 or more connected calls; the tooltip names which condition is missing |
+| Tool call evaluation is on but nothing gets scored | The agent is on Retell or Bland.ai, and [tool call evaluation only works for Vapi](/docs/simulation/guides/evaluate-tool-calls) | Switch the agent definition to Vapi if tool calls need scoring |
+| A chat run's SDK script finishes, but the run stays empty on the dashboard | `run_test_name` didn't match the run's name exactly, or `FI_API_KEY`/`FI_SECRET_KEY` are missing or wrong | Copy the name from the dashboard's boilerplate panel instead of retyping it, and confirm both keys are set |
+| `report.results` is always an empty list after `run_test` | Cloud mode doesn't populate that field; your results live on the dashboard, not in the SDK's return value | Read outcomes from the run's **Simulated runs**, **Call Details** (**Chat Details** on a chat run), and **Analytics** tabs, not the object `run_test` returns |
+| A call fails with a generic error instead of the exception your callback raised | The SDK reports a generic error to the dashboard rather than forwarding your exception's message | Log the exception yourself inside the callback if you need to know what actually went wrong |
+
+## Getting started
+
+**What do I need before I can run a simulation?**
+
+An [agent definition](/docs/simulation/concepts/agent-definitions) with at least one version, and a [scenario](/docs/simulation/concepts/scenarios) built against it with at least one eval attached. [Connect your agent](/docs/simulation/guides/connect-your-agent) and [Create scenarios](/docs/simulation/guides/create-scenarios) cover both.
+
+**Do I need a deployed agent to run anything?**
+
+Not for chat. [Simulate a prompt](/docs/simulation/guides/prompt-simulation) runs a saved Prompt Workbench version directly, with no agent definition and nothing to connect.
+
+**Do I need to write code?**
+
+Not for voice: Future AGI places the calls itself. A chat run needs a short Python callback using the `agent-simulate` SDK, which [Run a chat simulation](/docs/simulation/guides/run-chat-simulation) walks through.
+
+**Which voice providers are supported?**
+
+Vapi, Retell, and Bland.ai. See [Voice providers](/docs/simulation/reference/voice-providers) for what each one needs and supports, and use **Others** for an agent you can reach by phone.
+
+## Scenarios and generation
+
+**Why is a scenario greyed out when I try to select it for a run?**
+
+It has 0 rows. Open it and [add rows](/docs/simulation/guides/explore-scenarios/add-rows) before it can be picked.
+
+**What's the smallest and largest scenario I can generate?**
+
+Between 10 and 20,000 rows, whether you're generating a new scenario or [adding rows](/docs/simulation/guides/explore-scenarios/add-rows) to an existing one.
+
+**How many columns can I add at once?**
+
+Up to 10 in a single save from the [Add Column](/docs/simulation/guides/explore-scenarios/add-columns) drawer. Reopen it for more; there's no ceiling on the scenario itself.
+
+**Why did my dataset import get rejected?**
+
+A dataset needs at least 10 rows, no duplicate column names, and a `persona` column, if it has one, typed as Persona rather than plain text. The rejection names which condition failed.
+
+## Running a simulation
+
+**My chat run's Simulated runs tab is still empty. Is it broken?**
+
+No. A chat run stays empty until you run the SDK script yourself: it isn't queued anywhere and it doesn't time out. An empty tab means the script hasn't run, not that the run failed.
+
+**Why did my voice call cut off partway through?**
+
+Every call is capped at 30 minutes and ends automatically at that mark, regardless of provider. See [Call metrics](/docs/simulation/reference/call-metrics) for how a call's duration is reported once it completes.
+
+**Can I switch a run to a different voice provider after creating it?**
+
+No. The provider lives on the agent definition, not on the version, and both are fixed once a run test is created. Point a new run at a different agent definition instead. [Run a voice simulation](/docs/simulation/guides/run-voice-simulation) covers why the version matters more for voice than for chat.
+
+**Why does my chat conversation stop after 50 turns even though it shouldn't have ended yet?**
+
+50 turns is a safety cap that applies when a scenario's end condition never triggers. If conversations are cutting off early, check the scenario's flow rather than the agent.
+
+## Reruns and replay
+
+**Why does my chat run only offer "Run Evals", never "Run test + Evals"?**
+
+A chat agent's calls live in your own code, so the platform has nothing to replay; only its evals can rerun. Voice runs get both options, covered in [Run a voice simulation](/docs/simulation/guides/run-voice-simulation).
+
+**If a rerun fails right after I click it, did I lose the call's original data?**
+
+For **Run Evals**, no: a rerun that fails before it starts leaves the call exactly as it was. **Run test + Evals** is different: it clears the call's recording, transcript, and cost data as soon as it's dispatched, before the new call is placed, so a failure right after that point doesn't leave the original untouched. [Edit evals in a simulation](/docs/simulation/guides/edit-evals) covers what **Run Evals** changes and what it leaves untouched.
+
+**Does rerunning overwrite my previous result?**
+
+No. The prior state is kept so you can compare before and after. See [Runs & results](/docs/simulation/concepts/runs-and-results).
+
+**Why can't I replay this voice call's exact configuration?**
+
+Configuration replay only reconstructs Vapi calls. A Retell or Bland.ai call still replays, but you get a transcript comparison back rather than the original provider setup. [Replay voice calls](/docs/simulation/guides/replay-voice) has the detail.
+
+## Evals and tool calls
+
+**Why can't I delete this eval?**
+
+It's the last one on the run, and a run always needs at least one. [Add a replacement](/docs/simulation/guides/edit-evals) before removing it.
+
+**I turned on tool call evaluation, but nothing got scored.**
+
+Two possible reasons: the agent is on Retell, where tool call evaluation isn't wired at all, or the scenario simply never reached a tool call. [Evaluate tool calls](/docs/simulation/guides/evaluate-tool-calls) covers both.
+
+**Why don't tool calls show up in the transcript?**
+
+They're intentionally excluded from the transcript view. Check that call's tool-call results instead, alongside its other eval scores, covered in [Calls & transcripts](/docs/simulation/guides/explore-results/calls-and-transcripts).
+
+**My chat agent calls tools, but tool call evaluation still finds nothing to score.**
+
+The callback has to return an `AgentResponse` with `tool_calls` and `tool_responses` set, not a plain string. [Evaluate tool calls](/docs/simulation/guides/evaluate-tool-calls) has the exact shape.
+
+## Fix My Agent and optimization
+
+**Why is the Fix My Agent button disabled?**
+
+Two conditions, and the tooltip names which one is missing: the run has to be complete, and it needs at least 15 connected calls behind it. See [Fix My Agent](/docs/simulation/guides/fix-my-agent).
+
+**Fix My Agent says there are no suggestions. Is that an error?**
+
+No. Click refresh to run the analysis; if it genuinely finds nothing worth flagging, it says so rather than inventing an issue.
+
+**Does applying an optimization update my agent automatically?**
+
+No. An optimization run hands back a ranked list of trials and a best-performing prompt, but nothing pushes that prompt onto your agent for you. Copy the winning trial's text into a new agent version yourself, covered in [Optimization runs](/docs/simulation/guides/optimization-runs).
+
+## SDK
+
+**Which keys does the SDK need?**
+
+`FI_API_KEY` and `FI_SECRET_KEY`, read from your environment by `TestRunner()`, or passed directly as `api_key`/`secret_key`.
+
+**My keys look right, but the run stays empty. What's going on?**
+
+Missing credentials don't fail immediately, they only log a warning, then fail once the SDK actually calls the backend. Check both keys, and confirm `run_test_name` matches the run's name exactly.
+
+**Why does `report.results` come back empty even though the run completed?**
+
+That field isn't populated in cloud mode. Read outcomes from the run's **Simulated runs**, **Call Details** (**Chat Details** on a chat run), and **Analytics** tabs on the dashboard instead of the object `run_test` returns.
+
+**My callback raised an exception. Why does the dashboard just show a generic error?**
+
+The SDK reports a generic failure to the dashboard rather than your exception's message. Log it yourself inside the callback to see the real cause.
+
+**Can I return raw tool output without building `tool_calls`/`tool_responses` by hand?**
+
+Yes. Pass it through `metadata={"tool_outputs": [...]}` on your `AgentResponse` and the SDK converts it. [Evaluate tool calls](/docs/simulation/guides/evaluate-tool-calls) shows both paths.
+
+## Keep exploring
+
+
+
+ The four-step wizard that bundles an agent, scenarios, and evals
+
+
+ Where a run's calls, transcripts, and scores land
+
+
+ Turn a finished run's failures into a ranked list of fixes
+
+
+ Full field and method reference for agent-simulate
+
+
diff --git a/src/pages/index.astro b/src/pages/index.astro
index 8e92d422..d0f20457 100644
--- a/src/pages/index.astro
+++ b/src/pages/index.astro
@@ -15,7 +15,7 @@ const sections = [
href: "/docs",
links: [
{ title: "Installation", href: "/docs/installation" },
- { title: "Quickstart", href: "/docs/quickstart" },
+ { title: "Quickstart", href: "/docs/quickstart/setup-observability" },
]
},
{
@@ -47,8 +47,8 @@ const sections = [
color: "amber",
href: "/docs/simulation",
links: [
- { title: "Agent definition", href: "/docs/simulation/agent-definition" },
- { title: "Scenarios", href: "/docs/simulation/scenarios" },
+ { title: "Agent definition", href: "/docs/simulation/concepts/agent-definitions" },
+ { title: "Scenarios", href: "/docs/simulation/concepts/scenarios" },
]
},
{
@@ -69,8 +69,8 @@ const sections = [
color: "cyan",
href: "/docs/dataset",
links: [
- { title: "Create dataset", href: "/docs/dataset/create" },
- { title: "Experiments", href: "/docs/dataset/experiments" },
+ { title: "Create dataset", href: "/docs/dataset/guides/create-a-dataset" },
+ { title: "Experiments", href: "/docs/dataset/guides/run-an-experiment" },
]
},
{
@@ -81,8 +81,8 @@ const sections = [
href: "/docs/error-feed",
badge: "New",
links: [
- { title: "Using Google ADK", href: "/docs/error-feed/features/using-google-adk" },
- { title: "Taxonomy", href: "/docs/error-feed/taxonomy" },
+ { title: "Using Google ADK", href: "/docs/cookbook/error-feed/google-adk-multi-agent" },
+ { title: "Taxonomy", href: "/docs/error-feed/reference/error-taxonomy" },
]
},
{
@@ -92,8 +92,8 @@ const sections = [
color: "violet",
href: "/docs/optimization",
links: [
- { title: "Quickstart", href: "/docs/optimization/quickstart" },
- { title: "Bayesian search", href: "/docs/optimization/bayesian" },
+ { title: "Quickstart", href: "/docs/optimization/guides/run-an-optimization" },
+ { title: "Bayesian search", href: "/docs/optimization/reference/optimizers/bayesian-search" },
]
},
{
@@ -103,8 +103,8 @@ const sections = [
color: "yellow",
href: "/docs/prompt",
links: [
- { title: "Create from scratch", href: "/docs/prompt/create" },
- { title: "SDK integration", href: "/docs/prompt/sdk" },
+ { title: "Create from scratch", href: "/docs/prompt/guides/create-a-prompt" },
+ { title: "SDK integration", href: "/docs/prompt/reference/sdk-api" },
]
},
{
@@ -115,7 +115,7 @@ const sections = [
href: "/docs/agent-playground",
links: [
{ title: "Understanding Agent Playground", href: "/docs/agent-playground/concepts/understanding-agent-playground" },
- { title: "Build workflow", href: "/docs/agent-playground/features/build-workflow" },
+ { title: "Build workflow", href: "/docs/agent-playground/guides/build-workflow" },
]
},
{
@@ -126,7 +126,7 @@ const sections = [
href: "/docs/knowledge-base",
links: [
{ title: "Concepts", href: "/docs/knowledge-base/concept" },
- { title: "Create via SDK", href: "/docs/knowledge-base/sdk" },
+ { title: "Create via SDK", href: "/docs/knowledge-base/guides/manage-with-the-sdk" },
]
},
];
@@ -488,7 +488,7 @@ const colorMap: Record = {
-
+